{"as_of":"2026-08-05T16:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:65c38a273b76a46d656dc9129bc10b1c9405a90d505719baab86662d924105d8","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":32,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":32,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T11:19:51.012398Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":7,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"reference_index":170,"source":"pdf_text","source_observed_at":"2026-05-22T23:10:40.420241Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2406.04244"},"observation_digest":"sha256:dee4b4bb301e5b746ee9222eb87ac171a2060cfeea8af19b8d7f530bed43bce6","observation_id":"26dde097-e60d-47b9-ba7b-9ca5f4315064","resolution":{"observed_at":"2026-05-22T23:10:41.134008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"reference_index":208,"source":"pdf_text","source_observed_at":"2026-05-17T22:58:16.523267Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2406.11794"},"observation_digest":"sha256:a758c589fafcd06998f61b463b2ade72b8005d75afb47333462c9429b0409e5e","observation_id":"79ffe552-61e5-4673-bf9b-bc4fde5114c9","resolution":{"observed_at":"2026-05-17T22:58:17.334977Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-11T08:08:09.444352Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2406.12793"},"observation_digest":"sha256:510778642ccb379e633a0870e2262bdcd44ab0968f57dc4adf5b48a7a9585513","observation_id":"d04fd98e-06bd-41e4-af68-e6b519a0ec47","resolution":{"observed_at":"2026-05-11T08:08:09.653347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2507.22359","last_updated":"2026-04-14T11:47:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-30T03:50:46Z","title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","version":4},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-19T03:17:06.457421Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2507.22359"},"observation_digest":"sha256:f208a005a3d4048dd717c0ec2853cc63d20ccedf028c35ff28761508d4f0e93e","observation_id":"1bb22ce2-c68b-422d-948d-2f1f77c32e39","resolution":{"observed_at":"2026-05-19T03:22:01.341453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2509.20909","last_updated":"2026-05-09T19:01:47Z","snapshot_observed_at":"2026-08-02T22:05:17.866277Z","submitted_at":"2025-09-25T08:55:18Z","title":"LogitTrace: Detecting Benchmark Contamination via Layerwise Logit Trajectories","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-18T14:32:57.426996Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2509.20909"},"observation_digest":"sha256:5eaaf2458030f82207f5f6be2ae76d98e972040378d123b45706d4062aecbc20","observation_id":"54c99d7c-0932-4665-88df-4cafaffd7561","resolution":{"observed_at":"2026-05-18T14:36:28.943124Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-04T11:19:51.012398Z","title":"Rethinking benchmark and contam- ination for language models with rephrased samples, November 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.05709","last_updated":"2026-06-04T12:15:57Z","snapshot_observed_at":"2026-08-05T06:33:42.539704Z","submitted_at":"2025-10-07T09:22:22Z","title":"Correcting Prompt Dependence in LLM Benchmarks: A Bayesian Hierarchical Model with Embedding-Space Clustering","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-04T11:19:51.012398Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2510.05709"},"observation_digest":"sha256:f1081328abef3a2401976b44cd475de8ccc5c26d56301aee500d08b4f1b9932f","observation_id":"1cf0bda5-0b6e-4ddb-b0ff-0793c0142615","resolution":{"observed_at":"2026-08-04T11:19:51.012398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.05150","last_updated":"2026-04-06T20:25:20Z","snapshot_observed_at":"2026-07-06T22:53:55.926521Z","submitted_at":"2026-04-06T20:25:20Z","title":"Compiled AI: Deterministic Code Generation for LLM-Based Workflow Automation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T18:57:34.386632Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.05150"},"observation_digest":"sha256:18f68c5cfba9981a1898072c308578bbf1d94ba33c9597c3f5a6a0d521e8451d","observation_id":"d25e5c67-f333-4fda-b471-faf6e2b0bdff","resolution":{"observed_at":"2026-05-10T23:40:51.520975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.09251","last_updated":"2026-04-23T09:41:06Z","snapshot_observed_at":"2026-07-06T22:58:08.579543Z","submitted_at":"2026-04-10T12:07:22Z","title":"DRBENCHER: Can Your Agent Identify the Entity, Retrieve Its Properties and Do the Math?","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:46:24.780270Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.09251"},"observation_digest":"sha256:805b913da6e8ff10700cacb8af2a914e675b35b716a809cddf20cf1328d731d0","observation_id":"548a2693-0ac5-442b-aac0-89d3bec954ea","resolution":{"observed_at":"2026-05-11T08:11:04.832167Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.17966","last_updated":"2026-04-20T08:46:49Z","snapshot_observed_at":"2026-08-01T00:56:32.861823Z","submitted_at":"2026-04-20T08:46:49Z","title":"TPS-CalcBench: A Benchmark and Diagnostic Evaluation Framework for LLM Analytical Calculation Competence in Hypersonic Thermal Protection System Engineering","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-10T05:25:59.343884Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.17966"},"observation_digest":"sha256:dabfc65c14501244203a6ee18bb82620174b413bb3fd6e89cde375d72d34f9ac","observation_id":"47e2479c-fb1f-496d-827e-eaaf883b2921","resolution":{"observed_at":"2026-05-10T09:23:37.514127Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2604.24712","last_updated":"2026-04-27T17:21:09Z","snapshot_observed_at":"2026-07-06T23:10:42.926677Z","submitted_at":"2026-04-27T17:21:09Z","title":"When Prompt Under-Specification Improves Code Correctness: An Exploratory Study of Prompt Wording and Structure Effects on LLM-Based Code Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-08T03:00:26.137401Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2604.24712"},"observation_digest":"sha256:18cd0d1ea11c00d2e52b99d3e4d8d65cc5f500788a892e22bec016704403755e","observation_id":"22b7cd4d-8090-460d-9758-5f01a02de39b","resolution":{"observed_at":"2026-05-11T22:17:06.819446Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.02442","last_updated":"2026-05-04T10:42:26Z","snapshot_observed_at":"2026-07-30T10:52:38.632184Z","submitted_at":"2026-05-04T10:42:26Z","title":"Measuring AI Reasoning: A Guide for Researchers","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-08T18:53:18.586923Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.02442"},"observation_digest":"sha256:86d9391813e4bf8c727ca3b9b2e9b75f7e8508735e10d9bc2ca95e56d60ce3a4","observation_id":"6fd5b37e-9661-4f13-97f0-0af6e7d0b045","resolution":{"observed_at":"2026-05-09T06:05:36.959600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.04312","last_updated":"2026-05-05T21:24:58Z","snapshot_observed_at":"2026-07-06T23:17:04.200299Z","submitted_at":"2026-05-05T21:24:58Z","title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-08T17:06:32.814188Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.04312"},"observation_digest":"sha256:a3f8b1ad974c067538b946eb582ed26b4a3a89489a27d78e253fc682f8821775","observation_id":"1b15f2a6-e35a-4bde-a1ba-9bf71adfead8","resolution":{"observed_at":"2026-05-11T17:46:17.276350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.06327","last_updated":"2026-05-07T14:23:31Z","snapshot_observed_at":"2026-07-06T23:18:51.157048Z","submitted_at":"2026-05-07T14:23:31Z","title":"Measuring Evaluation-Context Divergence in Open-Weight LLMs: A Paired-Prompt Protocol with Pilot Evidence of Alignment-Pipeline-Specific Heterogeneity","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-08T10:23:02.697982Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.06327"},"observation_digest":"sha256:a8c4a46358384b5810065ed489f3db407f0f357e1ce32f69fdafe627c3969865","observation_id":"9d53cd78-ddd4-4c96-9571-3b76c9fa1010","resolution":{"observed_at":"2026-05-08T22:04:18.090953Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.07053","last_updated":"2026-05-26T07:47:31Z","snapshot_observed_at":"2026-08-02T22:34:50.237073Z","submitted_at":"2026-05-08T00:02:39Z","title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-11T00:57:55.616036Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.07053"},"observation_digest":"sha256:622c1952ccefb179e8f94aba1ce3eae7ca3d527a06c1dee980800e0225a68d23","observation_id":"1c5fe167-2a6f-4ffc-b264-cb3a266a33fe","resolution":{"observed_at":"2026-05-11T04:55:59.899239Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.07053","last_updated":"2026-05-26T07:47:31Z","snapshot_observed_at":"2026-08-02T22:34:50.237073Z","submitted_at":"2026-05-08T00:02:39Z","title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-30T23:42:00.965554Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.07053"},"observation_digest":"sha256:5eb425f2ed52eb27a53e5ae8e8ee82fefc33cc0bc245c5e14580e3907f507e43","observation_id":"bcaeda91-673a-474a-8963-2564111db7eb","resolution":{"observed_at":"2026-06-30T23:45:07.976186Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.10448","last_updated":"2026-05-11T12:20:15Z","snapshot_observed_at":"2026-08-04T21:52:34.354755Z","submitted_at":"2026-05-11T12:20:15Z","title":"Can Agent Benchmarks Support Their Scores? Evidence-Supported Bounds for Interactive-Agent Evaluation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-12T05:05:55.592359Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.10448"},"observation_digest":"sha256:a3cf3befcc0088bdb38a72360a0837c3ae31e51a88c0ff0bb593808b0d140a11","observation_id":"ec4241e4-e7c0-4b4c-8c06-4660896cd370","resolution":{"observed_at":"2026-05-12T05:41:23.900869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.11501","last_updated":"2026-05-12T04:21:26Z","snapshot_observed_at":"2026-07-06T23:23:21.034839Z","submitted_at":"2026-05-12T04:21:26Z","title":"Decaf: Improving Neural Decompilation with Automatic Feedback and Search","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T02:03:19.836506Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.11501"},"observation_digest":"sha256:1d2b494e2784717579bacf667fce3f3c4bcea9bc94b4f99970fe89974a0b309f","observation_id":"abfb013e-e714-4037-85ac-acc881089016","resolution":{"observed_at":"2026-05-13T02:07:08.093098Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-02T02:31:32.394072Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:b9b840409efd2f0fd5497b2e42af7133d83748f60077e910849851e331f8b9d4","observation_id":"481aa828-3a0e-498f-9f0b-f56df90e7622","resolution":{"observed_at":"2026-05-14T20:32:56.862291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.19999","last_updated":"2026-05-19T15:33:16Z","snapshot_observed_at":"2026-07-06T23:30:39.512029Z","submitted_at":"2026-05-19T15:33:16Z","title":"LLM Benchmark Datasets Should Be Contamination-Resistant","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-20T07:19:50.354875Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.19999"},"observation_digest":"sha256:f4f4bb4d2e01333d425430b2674677e99c507757809e9518c5dd71025f9ed5e7","observation_id":"a0f175ac-6597-48d7-9bf4-51fc3977f5ec","resolution":{"observed_at":"2026-05-20T07:23:07.058655Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.21543","last_updated":"2026-05-20T09:16:39Z","snapshot_observed_at":"2026-08-01T19:04:50.092578Z","submitted_at":"2026-05-20T09:16:39Z","title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","version":1},"reference_index":172,"source":"arxiv_source","source_observed_at":"2026-05-22T00:40:54.038367Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.21543"},"observation_digest":"sha256:d852e9faa603a14cc9371a4dfadd5fcfd29bc85e9c1af45b97a6ef4e3f3401e0","observation_id":"ae61037a-4a0c-4149-9764-5047fc27a22e","resolution":{"observed_at":"2026-05-22T00:44:29.489044Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.21856","last_updated":"2026-05-21T01:06:19Z","snapshot_observed_at":"2026-08-02T10:22:40.169491Z","submitted_at":"2026-05-21T01:06:19Z","title":"The Illusion of Reasoning: Exposing Evasive Data Contamination in LLMs via Zero-CoT Truncation","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-22T08:05:42.212459Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.21856"},"observation_digest":"sha256:0135a8a57b4128601df85383b01354d645f30adfa72a6f9f64e00f926a9a5bdd","observation_id":"e50e38d5-a330-4082-9fc1-b47897ecc449","resolution":{"observed_at":"2026-05-22T08:06:15.310194Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.23628","last_updated":"2026-05-22T13:40:00Z","snapshot_observed_at":"2026-07-06T23:33:49.013121Z","submitted_at":"2026-05-22T13:40:00Z","title":"How Hard is it to Rig a Benchmark? A Social Choice Analysis of Leaderboard Robustness","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-25T04:37:01.537128Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.23628"},"observation_digest":"sha256:2fa63aa7cf608d41506385f74051b96e8d069bd105dfa551dc11baefb99e3d0f","observation_id":"81a8cf3c-06b8-49b8-bbd7-3da728bad548","resolution":{"observed_at":"2026-05-25T04:40:24.444660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24079","last_updated":"2026-05-22T17:30:20Z","snapshot_observed_at":"2026-08-01T09:58:43.814751Z","submitted_at":"2026-05-22T17:30:20Z","title":"TRACER: A Semantic-Aware Framework for Fine-Grained Contamination Detection in Code LLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-30T15:32:05.976335Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24079"},"observation_digest":"sha256:946baf5798622ed586b82fec5a9504b5109f1cb39409645123968229904339f0","observation_id":"e11c3d40-017b-429b-8690-ff2370e45d98","resolution":{"observed_at":"2026-06-30T15:34:47.895008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24213","last_updated":"2026-05-22T20:54:30Z","snapshot_observed_at":"2026-08-02T06:58:01.300783Z","submitted_at":"2026-05-22T20:54:30Z","title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-30T14:41:07.354007Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24213"},"observation_digest":"sha256:fcfa2db970d28f8edc6e31c51141c5e100e94dce3307a755eb8fcb7d410dbb4c","observation_id":"04ea9e21-1c59-4575-a510-f812fd2b7aeb","resolution":{"observed_at":"2026-06-30T14:44:45.155954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24661","last_updated":"2026-07-01T08:31:40Z","snapshot_observed_at":"2026-08-03T01:04:22.766287Z","submitted_at":"2026-05-23T17:03:42Z","title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-30T13:27:50.367497Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24661"},"observation_digest":"sha256:8b6961d3ace60ac60dfa382d16eafec3bc26ec6600ec4e3c1bba9f58e00a6e3b","observation_id":"6c2dc137-5f7d-4787-b49a-71cbffd49ef8","resolution":{"observed_at":"2026-06-30T13:34:40.588543Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24661","last_updated":"2026-07-01T08:31:40Z","snapshot_observed_at":"2026-08-03T01:04:22.766287Z","submitted_at":"2026-05-23T17:03:42Z","title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-01T07:35:51.017797Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24661"},"observation_digest":"sha256:272068cc0e5e37eca159bea03783d120849679f6ed50d5873e550b0889f787fe","observation_id":"b5a659b1-05df-4a48-8f38-913a67b3cfe8","resolution":{"observed_at":"2026-07-01T08:05:31.914578Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.24661","last_updated":"2026-07-01T08:31:40Z","snapshot_observed_at":"2026-08-03T01:04:22.766287Z","submitted_at":"2026-05-23T17:03:42Z","title":"Measuring Reasoning Quality in LLMs: A Multi-Dimensional Behavioral Framework","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-02T23:22:08.567841Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.24661"},"observation_digest":"sha256:2488b67e7a6870261d3bfdeb5fc60a79b97300d30160ca601db7e27d8a752054","observation_id":"73c90e48-132c-4c28-9ee1-a4949e9206f0","resolution":{"observed_at":"2026-07-02T23:27:26.731890Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2605.26161","last_updated":"2026-05-24T14:59:12Z","snapshot_observed_at":"2026-08-02T04:25:25.218976Z","submitted_at":"2026-05-24T14:59:12Z","title":"TSFMAudit: Data Contamination Auditing in Forecasting Time Series Foundation Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-30T12:00:05.807004Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2605.26161"},"observation_digest":"sha256:146d21bdc274c1122e6bb9b7d56c59d355edee89855b1259a32f11f5175221a7","observation_id":"dab2454a-666d-4842-b07a-aed17edf3fd0","resolution":{"observed_at":"2026-06-30T12:04:38.780577Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2606.11909","last_updated":"2026-06-10T10:37:27Z","snapshot_observed_at":"2026-07-06T23:50:53.469137Z","submitted_at":"2026-06-10T10:37:27Z","title":"Embodied-BenchClaw: An Autonomous Multi-Agent System for Embodied Spatial Intelligence Benchmark Construction","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T09:48:49.786021Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2606.11909"},"observation_digest":"sha256:39b46454424bee37eaa612f5f550579e9ef98118bb84d4318c503a27e854e0d7","observation_id":"429ec9d5-66e8-4def-a403-8144313ea449","resolution":{"observed_at":"2026-07-03T10:48:02.880656Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2606.12385","last_updated":"2026-06-10T17:47:59Z","snapshot_observed_at":"2026-08-05T09:55:05.283277Z","submitted_at":"2026-06-10T17:47:59Z","title":"Which Models Are Our Models Built On? Auditing Invisible Dependencies in Modern LLMs","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T09:57:14.328157Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2606.12385"},"observation_digest":"sha256:40079312286dcf4667dbb10c78818f84aafd8f97072393d36bf3b5d2c03c1df0","observation_id":"dcbed4ea-62a8-48a1-b727-be0d3cb6004e","resolution":{"observed_at":"2026-07-03T10:37:56.593959Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":"2311.04850","doi":"10.48550/arxiv.2311.04850","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Guanhua Zhang and Moritz Hardt","venue":"arXiv (Cornell University)","work_id":"76543ba2-5ea1-419d-aad3-b610c7403b0f","year":2023},"citing_paper":{"arxiv_id":"2606.29815","last_updated":"2026-06-29T05:48:42Z","snapshot_observed_at":"2026-08-01T15:01:23.289070Z","submitted_at":"2026-06-29T05:48:42Z","title":"SrDetection: A Self-Referential Framework for Data Leakage Detection in Code Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-30T06:20:02.046020Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2606.29815"},"observation_digest":"sha256:134d6672d3448c9f48b6d7c406b52a76f690d17310a2302046bf14400dff0006","observation_id":"df54faea-0bd2-4570-a54c-c347743948f7","resolution":{"observed_at":"2026-06-30T06:24:18.367339Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-01T05:03:42.877641Z","title":"arXiv preprint arXiv:2311.04850 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22368","last_updated":"2026-07-24T14:55:19Z","snapshot_observed_at":"2026-08-01T05:03:37.778828Z","submitted_at":"2026-07-24T14:55:19Z","title":"Do Agent Benchmarks Measure Capability? Protocol Validity in the Age of Agentic AI","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-01T05:03:42.877641Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2607.22368"},"observation_digest":"sha256:e07df4983b2f28677f5dea2ce6d46102304e02f4ffda012a041169ba59159105","observation_id":"e34d5daa-ff82-4dcf-8c9b-19274d08c5d6","resolution":{"observed_at":"2026-08-01T05:03:42.877641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.04850/citation-record","integrity":"/paper/2311.04850/integrity","json":"/paper/2311.04850/citation-record.json","paper":"/paper/2311.04850"},"outbound":[],"paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-04T02:43:33.877060Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 32 inbound Pith citation observations for arXiv:2311.04850."}