{"as_of":"2026-08-07T02:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bd68a52868b658893911850acc62d7623e88ab0850af79a8ca67799552db0d49","coverage":[{"denominator":26,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T17:13:04.305435Z","state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T11:29:13.728227Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-11T09:00:59.794538Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"cited_work":{"arxiv_id":"2604.07650","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.07650","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","venue":"cs.AI","work_id":"aa81c8ec-ccdd-4315-83e5-1a87101b6dcf","year":2026},"citing_paper":{"arxiv_id":"2604.12185","last_updated":"2026-04-14T01:31:35Z","snapshot_observed_at":"2026-07-06T23:00:28.036409Z","submitted_at":"2026-04-14T01:31:35Z","title":"Knowledge Is Not Static: Order-Aware Hypergraph RAG for Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T16:20:31.981340Z"},"links":{"cited_paper":"/paper/2604.07650","citing_paper":"/paper/2604.12185"},"observation_digest":"sha256:6931e6cd8f477bc9d64e2202823933898a2bc21644d588dfe1dc1c17da6173cf","observation_id":"a544d1b3-1055-439b-ac59-126ffae119ee","resolution":{"observed_at":"2026-05-11T09:00:59.799891Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.07650","snapshot_observed_at":"2026-08-01T11:29:13.728227Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19899","last_updated":"2026-07-22T08:35:00Z","snapshot_observed_at":"2026-08-04T09:33:45.861093Z","submitted_at":"2026-07-22T08:35:00Z","title":"Harnessing Disagreement: Detecting Correlated Agreement Blindness in Multi-Agent Triage","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T11:29:13.728227Z"},"links":{"cited_paper":"/paper/2604.07650","citing_paper":"/paper/2607.19899"},"observation_digest":"sha256:822d5f102e55f91f2dd53409db72289cf9d710cdb995dfea1cc4f12ea62a5b52","observation_id":"4e42e32a-aaa7-4861-80d4-4c3e1a07cccd","resolution":{"observed_at":"2026-08-01T11:29:13.728227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.07650","snapshot_observed_at":"2026-07-31T14:04:25.242442Z","title":"arXiv preprint arXiv:2604.07650 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24471","last_updated":"2026-07-27T14:07:12Z","snapshot_observed_at":"2026-08-06T02:49:03.228120Z","submitted_at":"2026-07-27T14:07:12Z","title":"Grounding latent algorithm routing in transformer reasoning","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-31T14:04:25.242442Z"},"links":{"cited_paper":"/paper/2604.07650","citing_paper":"/paper/2607.24471"},"observation_digest":"sha256:ae520c63065f201c8c411376465fb6ab82691799f2566dfb5f30ddcfb4981d33","observation_id":"22513b24-f95b-439b-8871-aae44cf9808d","resolution":{"observed_at":"2026-07-31T14:04:25.242442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.07650/citation-record","integrity":"/paper/2604.07650/integrity","json":"/paper/2604.07650/citation-record.json","paper":"/paper/2604.07650"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.10925","last_updated":"2025-08-08T19:24:38Z","snapshot_observed_at":"2026-08-01T16:27:35.664983Z","submitted_at":"2025-08-08T19:24:38Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","version":1},"cited_work":{"arxiv_id":"2508.10925","doi":"10.3115/1073083.1073135","metadata_source":"pith","pith_arxiv_id":"2508.10925","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","venue":"cs.CL","work_id":"178c1f7e-4f19-4392-a45d-45a6dfa88ead","year":2025},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2508.10925","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:808b8191ae6fafac8fb2d1cf0a4af304f5ed69ba27a9ca035628c2b4cc00b46b","observation_id":"45ccfa34-725b-414e-8136-939cb84314d4","resolution":{"observed_at":"2026-05-11T07:20:58.913602Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T19:19:47.962081+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T19:19:47.962081+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"anthropic.com/news/claude-3-family","venue":null,"work_id":"c23690e5-8b9a-4384-8e8b-0556ad091843","year":2026},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:7e41f4e13d4d7c3f363efa44c524644c72ca0cc7c73935d7cafd0b2c0c6e3b08","observation_id":"be98a279-555f-4c17-ac85-36eacbdc39c2","resolution":{"observed_at":"2026-05-17T11:31:46.606013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:cee9735829a880ea511eb82535025b88817cf6e16398ff020b19620b4001046c","observation_id":"2cb65387-d213-4111-bdaa-2acbc2c3cffc","resolution":{"observed_at":"2026-05-11T07:20:58.955929Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Beyond the surface: Measuring self-preference in llm judgments","venue":null,"work_id":"6ad8ade4-bfc7-43ef-a4f8-efdef928e9f1","year":2025},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:df667e0c2469a632867c90fa162c8168a5b34153e256f3d53470fafc56e14744","observation_id":"05dfea4b-ad56-4640-9520-b4d45d004aa2","resolution":{"observed_at":"2026-05-17T11:31:46.612265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Investigat- ing data contamination in modern benchmarks for large language models","venue":null,"work_id":"0be91c26-4407-42cf-b992-154ce48b3eb8","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:5fd182aced6f95183b744d70dee53f3512d3812a7b2e939ef32733b57cfef062","observation_id":"5e083a0e-f225-4d4c-abcc-945affb1a190","resolution":{"observed_at":"2026-05-17T11:31:46.619461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gen- eralization or memorization: Data contamination and trustworthy evaluation for large language models","venue":null,"work_id":"a1913c6b-1693-4070-9921-f2e8afe4681f","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:7581da3fe235f626a04e12209b974211095718fe0d1be37ffe72d7086b1fa169","observation_id":"08c57516-465f-4a13-9bd6-b1dd58f6c11d","resolution":{"observed_at":"2026-05-17T11:31:46.589427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:67c566c9fd52a42042b595c130a57dd5fa4e7dc0f3f89458e9c9d2ff669534a6","observation_id":"9b32f6f8-9314-4fa8-b5da-cb107f2ce88e","resolution":{"observed_at":"2026-05-11T07:20:58.959742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:f203524ac31666d1b596aa9cd4e310b71915514318fac4de93039f7953c46e97","observation_id":"d6d876d2-44ea-4f21-ab4c-5c99360345aa","resolution":{"observed_at":"2026-05-11T07:20:58.857547Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.03253","last_updated":"2023-05-05T02:46:22Z","snapshot_observed_at":"2026-08-05T18:08:06.664474Z","submitted_at":"2023-05-05T02:46:22Z","title":"VicunaNER: Zero/Few-shot Named Entity Recognition using Vicuna","version":1},"cited_work":{"arxiv_id":"2305.03253","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.03253","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vicunaner: Zero/few-shot named entity recognition using vicuna.arXiv preprint arXiv:2305.03253","venue":null,"work_id":"812095a8-fdcf-4591-89b4-d2f97cda055c","year":null},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2305.03253","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:2873ff5feb343195810b226920aaac7c3081e125e89f0510df5b587d0ad3d217","observation_id":"aba21ea2-547c-4c03-946f-c625043e6dd7","resolution":{"observed_at":"2026-05-11T07:20:58.963826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.01534","doi":"10.48550/arxiv.2502.01534","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2502.01534 , year=","venue":"arXiv (Cornell University)","work_id":"ac54fe1d-85a4-49da-a17f-ca219afb9abe","year":2025},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:d28272cd26ab47b2b31cebf2e1e9a049a57def817013d43bcb9ffa675327822a","observation_id":"135829f9-52a7-4128-a77b-d841e8affac8","resolution":{"observed_at":"2026-05-11T07:20:58.892444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04434","last_updated":"2024-06-19T06:04:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","version":5},"cited_work":{"arxiv_id":"2405.04434","doi":"10.1145/3593013.3594097","metadata_source":"pith","pith_arxiv_id":"2405.04434","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","venue":"cs.CL","work_id":"1e1df141-cac8-47fd-b068-c4c96e51e331","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2405.04434","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:2f6f0ca14aed69f0508d576bfe4b47a86354e416f3fe323493c529b536b62be4","observation_id":"0e1387c6-a6b4-487d-82fa-df1282b8a272","resolution":{"observed_at":"2026-05-11T07:20:58.887342Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accessed: 2026-03-31","venue":null,"work_id":"d58ed9df-088f-43a5-9b62-1658d99c530c","year":2026},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:a1af8825dc1f0e9c3de81b0cf3d117877eb66eb7b660ee874cbdfec11be4de7a","observation_id":"f93f71fb-5673-46c0-80a8-bc6a68fe120f","resolution":{"observed_at":"2026-05-17T11:31:46.615373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wonpyo Park, Dongju Kim, Yan Lu, and Minsu Cho","venue":null,"work_id":"a0134011-20de-4be6-9f4e-f9dcbbdfbb6a","year":2026},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:5d9c5886ed92423b53173c37b0f9585903e794d56e426b2a5a2dc88e67eab7ad","observation_id":"5e3121e0-dd79-4c48-b006-410650061054","resolution":{"observed_at":"2026-05-17T11:31:46.579752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":"2304.03277","doi":"10.48550/arxiv.2304.03277","metadata_source":"pith","pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction Tuning with GPT-4","venue":"cs.CL","work_id":"fd515477-f9f1-48aa-9feb-a3308e7656bb","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:18d0f1604aa72643bc59743875584fb5def80d1522654cd88b912cc86e799971","observation_id":"a8509044-21a0-43fe-acbc-a89901447a6d","resolution":{"observed_at":"2026-05-14T17:04:18.193148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":null,"work_id":"892d1463-0841-4a79-86f0-f4366983a3bd","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:6d42f3cd05433490354f2e87d5251bc0ebbc5adda74af3023a5ffced1691b27a","observation_id":"ea98a32b-8e6f-4555-8ef5-a0dc014494af","resolution":{"observed_at":"2026-05-17T11:31:46.583568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16789","last_updated":"2024-03-09T22:26:06Z","snapshot_observed_at":"2026-07-06T16:38:29.188225Z","submitted_at":"2023-10-25T17:21:23Z","title":"Detecting Pretraining Data from Large Language Models","version":3},"cited_work":{"arxiv_id":"2310.16789","doi":"10.48550/arxiv.2310.16789","metadata_source":"pith","pith_arxiv_id":"2310.16789","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Detecting Pretraining Data from Large Language Models","venue":"cs.CL","work_id":"1ff0530f-0b29-487b-ba43-d22a740293b1","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2310.16789","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:6556598b31b0bc19ec26f451ac9b3a23459b19b55dd5be192f6fa07f639ab776","observation_id":"e6a2227d-c024-43ef-906f-6b345c260a27","resolution":{"observed_at":"2026-05-17T18:07:25.061786Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":"2601.03267","doi":"10.48550/arxiv.2601.03267","metadata_source":"pith","pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"OpenAI GPT-5 System Card","venue":"cs.CL","work_id":"ca87689a-0d29-4476-b504-b65dbbb08af4","year":2025},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:1e6cf97b15df7cf98605c5e70e29458365f0248018ad0c57f1d544ab641b16b2","observation_id":"4cbde1b0-05ca-4a33-8170-13613f8e2f96","resolution":{"observed_at":"2026-05-11T07:20:58.948246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:8cdbef6e315dc004bcc31d909cb6a9f1591618f26fce269db1ac4070c30b8b09","observation_id":"5e5771af-2a38-4203-88b1-f67c0b69f26d","resolution":{"observed_at":"2026-05-11T07:20:58.872920Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":"2307.09288","doi":"10.24963/ijcai.2025/706","metadata_source":"pith","pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","venue":"cs.CL","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:8a9cb6eed102bc55af543c296ce4c84eb1b97e011777810cc3abb0576c43c016","observation_id":"bc23bf87-020e-4918-b012-c9bee0b681a5","resolution":{"observed_at":"2026-05-11T07:20:58.944832Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Who taught you that? tracing teachers in model distillation","venue":null,"work_id":"b8a3f8b6-e765-46c7-87d9-fd60838aafe4","year":2025},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:c2168e254a065262ddc85ddc650fb4a36e76af30da361b5bb1920e226a09dc97","observation_id":"c22ab2d3-dc15-4fa6-8b31-c76c8a472981","resolution":{"observed_at":"2026-05-17T11:31:46.602467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05087","last_updated":"2024-05-24T06:37:31Z","snapshot_observed_at":"2026-07-06T15:40:12.711962Z","submitted_at":"2023-06-08T10:41:56Z","title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","version":2},"cited_work":{"arxiv_id":"2306.05087","doi":"10.48550/arxiv.2306.05087","metadata_source":"pith","pith_arxiv_id":"2306.05087","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pandalm: An automatic evaluation benchmark for llm instruction tuning optimization","venue":"cs.CL","work_id":"e19da613-b037-4e35-b0c9-c0024fcb762e","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2306.05087","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:c54e88ba7d7659fbef25a3c91c1e7c6d757fbfba384f97acd6920d80865a93e3","observation_id":"dec08c7f-4854-4589-bb93-55abe983f543","resolution":{"observed_at":"2026-05-11T07:20:58.903293Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21819","last_updated":"2025-06-21T08:39:06Z","snapshot_observed_at":"2026-08-04T18:30:37.623897Z","submitted_at":"2024-10-29T07:42:18Z","title":"Self-Preference Bias in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":"2410.21819","doi":"10.48550/arxiv.2410.21819","metadata_source":"pith","pith_arxiv_id":"2410.21819","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-Preference Bias in LLM-as-a-Judge","venue":"cs.CL","work_id":"b98a1723-3323-42b3-a741-8b6785806d2d","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2410.21819","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:fd7386c8f97cf20f9f3b3fc18d962380bfd2d629fdc483f6d1851c07da2775e0","observation_id":"ddf47d33-7cd9-4f04-934d-022046108481","resolution":{"observed_at":"2026-05-15T14:46:32.590484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":"2406.04244","doi":"10.48550/arxiv.2406.04244","metadata_source":"pith","pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","venue":"cs.CL","work_id":"30fe188d-51ed-4ae5-8557-dfd5c814931e","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:e4c63031a1c1c167b689b81b8cca427d96bca998177085a6edc3a06686d9366c","observation_id":"ed014e5e-1fe5-4460-9cdd-94128788d928","resolution":{"observed_at":"2026-05-22T23:10:41.376209Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02736","last_updated":"2024-10-04T03:57:47Z","snapshot_observed_at":"2026-08-01T08:21:19.528254Z","submitted_at":"2024-10-03T17:53:30Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":"2410.02736","doi":"10.48550/arxiv.2410.02736","metadata_source":"pith","pith_arxiv_id":"2410.02736","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge","venue":"cs.CL","work_id":"226fa5a3-d2b0-46fc-b73d-cfb5ca7b6a4c","year":2024},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2410.02736","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:7b33f9da1df546202e7f273f44d0c6ec54862d2131cf191becc56a0ec7d6aaba","observation_id":"2f074eef-00b6-4cf4-88dd-c85cc68b3796","resolution":{"observed_at":"2026-05-15T20:00:24.757706Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:49:47.157032+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:49:47.157032+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01964","last_updated":"2023-11-03T14:59:54Z","snapshot_observed_at":"2026-08-06T10:56:05.075839Z","submitted_at":"2023-11-03T14:59:54Z","title":"Don't Make Your LLM an Evaluation Benchmark Cheater","version":1},"cited_work":{"arxiv_id":"2311.01964","doi":"10.48550/arxiv.2311.01964","metadata_source":"pith","pith_arxiv_id":"2311.01964","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Don't make your llm an evaluation benchmark cheater","venue":"cs.CL","work_id":"7ea5a199-3c68-4578-a9b4-22b8573dcaa3","year":2023},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"cited_paper":"/paper/2311.01964","citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:9983adbc2928a0ec7844d177c8e6a10df99533fcf87f291270b9e82b7cb940d8","observation_id":"227c3cac-82a2-40f0-be47-0d92e3231ded","resolution":{"observed_at":"2026-05-11T07:20:58.932401Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Instruct","venue":null,"work_id":"c3a8e29c-6bff-463f-a9cd-cc93f03ba2c3","year":2025},"citing_paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T17:13:04.305435Z"},"links":{"citing_paper":"/paper/2604.07650"},"observation_digest":"sha256:0fd57a8fdfb16f4707d77fdd4e574eaf6924f0a1819492ea91c5b900a6533250","observation_id":"dda0d095-5306-4d66-8f7a-4072d4d5d2e6","resolution":{"observed_at":"2026-05-17T11:31:46.597778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.07650","last_updated":"2026-04-08T23:32:06Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-02T20:20:50.082823Z","submitted_at":"2026-04-08T23:32:06Z","title":"How Independent are Large Language Models? A Statistical Framework for Auditing Behavioral Entanglement and Reweighting Verifier Ensembles"},"reference_resolution":{"displayed":26,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":16,"verified_fuzzy":9},"total_outbound_references":26},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 26 of 26 outbound references and 3 inbound Pith citation observations for arXiv:2604.07650."}