{"as_of":"2026-08-07T22:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1c93362af4f7a9ab9dfde46b9e94c7ae8dd326f28e0a22d45ef279a24a34f362","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:44:09.471855Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.17747/citation-record","integrity":"/paper/2507.17747/integrity","json":"/paper/2507.17747/citation-record.json","paper":"/paper/2507.17747"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.907090Z","title":"Claude 3.5 sonnet","venue":null,"work_id":"39614098-eb34-40e7-91e9-d720b30d0a80","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.233992Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c48dc02933c4a1261e6463a662d91c44f3b905ed2737f27b672baf3d09cb1243","observation_id":"a83d6f7c-cc0d-4c07-9845-7e11451947d1","resolution":{"observed_at":"2026-08-06T14:44:11.909879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.898638Z","title":"Claude 3.5 haiku","venue":null,"work_id":"fdf37974-ef36-4f02-a750-35a5fc67dd4c","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.237996Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:4cbb35737f2153ea8f0fa38f2ff8f720deda6065fc8d39bc1898ac0b1c2ea8e9","observation_id":"0c351428-9b2f-4c72-bcd6-0b4c6cf93302","resolution":{"observed_at":"2026-08-06T14:44:11.901402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.890035Z","title":"ARC-AGI-2 + ARC Prize 2025 is Live! Blog Post, https://arcprize.org/blog/announcing-arc-agi-2-and-arc-prize-2025, March 24 2025","venue":null,"work_id":"0081dad1-40b5-4b5b-94f5-e666ff04447d","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.241713Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:40782715bde0ec146a5f9b879803716d7c9569d29eefa26c05546b1484c11a56","observation_id":"00efc9cd-aaad-4095-8e28-dba4fd18b6a8","resolution":{"observed_at":"2026-08-06T14:44:11.893059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04181","last_updated":"2023-11-04T11:50:13Z","snapshot_observed_at":"2026-07-06T15:39:33.825891Z","submitted_at":"2023-06-07T06:29:58Z","title":"Benchmarking Foundation Models with Language-Model-as-an-Examiner","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.04181","snapshot_observed_at":"2026-08-06T14:44:09.245058Z","title":"Benchmarking foundation models with language-model-as-an-examiner","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.245058Z"},"links":{"cited_paper":"/paper/2306.04181","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:f92c8acbee74c2fe926b9fccbad573a2707d2a8cba94d134195a89530fd66637","observation_id":"168020df-faa3-4617-bcb0-3f661ff0428e","resolution":{"observed_at":"2026-08-06T14:44:09.245058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.881334Z","title":"Leak, cheat, repeat: Data contamination and evaluation malpractices in closed-source llms","venue":null,"work_id":"4c53be42-aea0-4a7a-8ef8-a33413de120b","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.248532Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e1cf3cd16291d027374e9fb069d7b3762e4c8bc6ee7cffccfc1b0d7f5d3f351c","observation_id":"d2899a6c-dd33-44ef-9b7f-4fa173d23ace","resolution":{"observed_at":"2026-08-06T14:44:11.884260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.252377Z","title":"Adversarial multi-agent evaluation of large language models through iterative debates","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.252377Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:6be0c2a24186e98cff8af327f05b57efd1e438f80c465a0b06076d72746a8925","observation_id":"8e75f131-7bc5-43fa-b6b7-b20bf3ce0b61","resolution":{"observed_at":"2026-08-06T14:44:09.252377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.872613Z","title":"Flageval","venue":null,"work_id":"9bc0cb3e-4ecc-4244-9cee-7778df2f4bda","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.255787Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c6c2ca554c9cc06206bae72d11dace6fe38849e3d0934234e89c2f410edcc471","observation_id":"325cfa56-0309-4fc2-bdd2-eb2b2dc9dbb9","resolution":{"observed_at":"2026-08-06T14:44:11.875447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13408","last_updated":"2025-05-19T17:44:26Z","snapshot_observed_at":"2026-08-07T15:43:00.197948Z","submitted_at":"2025-05-19T17:44:26Z","title":"CoT-Kinetics: A Theoretical Modeling Assessing LRM Reasoning Process","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13408","snapshot_observed_at":"2026-08-06T14:44:09.258694Z","title":"Cot-kinetics: A theoretical modeling assessing lrm reasoning process, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.258694Z"},"links":{"cited_paper":"/paper/2505.13408","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:226cb92c75d490275e9ca9aff8e1f868b1d52fdaf0efb296bc15e2c9e82c6a29","observation_id":"28e252be-bc05-477b-9b7c-6b1eb1892a54","resolution":{"observed_at":"2026-08-06T14:44:09.258694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.261955Z","title":null,"venue":null,"work_id":null,"year":1952},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.261955Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:602872e2552bef8d66455aab6dab893380ec69b721c912f4c5b5a0bad5298b35","observation_id":"91128ddb-5cc4-4299-be72-ffd00b990fdd","resolution":{"observed_at":"2026-08-06T14:44:09.261955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-06T14:44:09.264731Z","title":"Sparks of artificial general intelligence: Early experiments with gpt-4, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.264731Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:40258123152564922cf2ddab460bb5eece5e3446e9bbee899d6d81b36c48bc24","observation_id":"f00b4b03-06ee-4086-9adc-237c2aaeaa5e","resolution":{"observed_at":"2026-08-06T14:44:09.264731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02892","last_updated":"2025-07-07T23:41:53Z","snapshot_observed_at":"2026-07-06T19:27:16.301809Z","submitted_at":"2024-10-03T18:30:47Z","title":"The Role of Deductive and Inductive Reasoning in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02892","snapshot_observed_at":"2026-08-06T14:44:09.268106Z","title":"The role of deductive and inductive reasoning in large language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.268106Z"},"links":{"cited_paper":"/paper/2410.02892","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:b8f0c3309f11eb9fd0f9ba2e3979b4db9c35f43efd253fb02af5f46c65f11371","observation_id":"d8b5ddd1-7732-46aa-a3de-03b2b08f4d5c","resolution":{"observed_at":"2026-08-06T14:44:09.268106Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.863689Z","title":"Are we on the right way for evaluating large vision-language models? In A","venue":null,"work_id":"643f7f49-aa94-453d-b1ae-b0caf0356341","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.271284Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:9a801dbbc54988ec79a0317525e1195506d6166880a5aea3277cff9d14e11a46","observation_id":"9dca74ff-0d91-4e32-ae9c-e9c730ce254e","resolution":{"observed_at":"2026-08-06T14:44:11.866662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.855413Z","title":"Jordan, Joseph E","venue":null,"work_id":"1d56a819-015d-471d-b8de-60984a907b2f","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.274046Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:86cc9de05b832ff37dbae2fc098bdda404226f274031bc8aaf5517b67e008428","observation_id":"2000b724-3372-452c-975c-a70ce8d73640","resolution":{"observed_at":"2026-08-06T14:44:11.858241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04604","last_updated":"2025-01-08T05:24:50Z","snapshot_observed_at":"2026-08-04T14:31:04.378756Z","submitted_at":"2024-12-05T20:40:28Z","title":"ARC Prize 2024: Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04604","snapshot_observed_at":"2026-08-06T14:44:09.277264Z","title":"Arc prize 2024: Technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.277264Z"},"links":{"cited_paper":"/paper/2412.04604","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:666c359d6cc4c45c2c64b59f33f69eed670b052dfb720e8db92069ba51e4c271","observation_id":"8877c7e4-7337-4901-9ce7-33441fd4fad0","resolution":{"observed_at":"2026-08-06T14:44:09.277264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.846483Z","title":"Abstraction and reasoning corpus for artificial general intelligence (arc-agi), 2019","venue":null,"work_id":"75668d08-3ace-451b-9943-ec9cd5553266","year":2019},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.280482Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:50c48dbccea2c9760f9603ddf65828b6df5aa4353cd810815f477b86d0a06fdd","observation_id":"8e8b6024-00f3-460a-a420-2b2ccca90a02","resolution":{"observed_at":"2026-08-06T14:44:11.849645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T14:44:09.283458Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.283458Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:129153a7bd520f46deda3644987caff9c0795ea3e3c340acd6c345371c0edcb7","observation_id":"85429f59-fd84-4ab2-a982-cdf409324331","resolution":{"observed_at":"2026-08-06T14:44:09.283458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T14:44:09.286882Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.286882Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ea78a7c9b53435a816f790625fa6ab8f4f0cbabe909243f63164edc487605dbf","observation_id":"48812bb5-00e1-424c-b0e9-4dafced156f7","resolution":{"observed_at":"2026-08-06T14:44:09.286882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T14:44:09.289691Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.289691Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:58c26d85ba3189d14c4024fee686f8a236598b78e555c9241861b6b5980ea808","observation_id":"a3ba679b-47be-4e9a-b1ff-285680b552b7","resolution":{"observed_at":"2026-08-06T14:44:09.289691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.292859Z","title":"Investigating data contamination in modern benchmarks for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.292859Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:f1a90db1e6dc9052e96aff010012de0478456d515348b21ce7da81671354ed0d","observation_id":"ed6de936-dfbd-410d-b224-bb625d1ae89a","resolution":{"observed_at":"2026-08-06T14:44:09.292859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.836932Z","title":"Stabilizing modality gap & lowering gradient norms improve zero-shot adversarial robustness of vlms","venue":null,"work_id":"5833d507-db24-4699-a81c-3064f07cf1de","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.295646Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:67f1aafab91d92f47776983aff2a4a36768ceecadf56a3e4d7c3bc3bf0b74f8b","observation_id":"a740c12f-d3a0-49c5-b8d7-ef683acdcc95","resolution":{"observed_at":"2026-08-06T14:44:11.840252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.827180Z","title":"Improving zero-shot adversarial robustness in vision-language models by closed-form alignment of adversarial path simplices","venue":null,"work_id":"3c215420-7b39-409d-b7f2-e2d406928138","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.298363Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:eb85e55804a94877a946ffea8e46fbc9ff7fbf0404ab4865fecdebbc443d95c5","observation_id":"8f4b08d6-f95c-48ea-9b21-134bb8dd3aba","resolution":{"observed_at":"2026-08-06T14:44:11.830837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14325","last_updated":"2023-05-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T17:55:11Z","title":"Improving Factuality and Reasoning in Language Models through Multiagent Debate","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14325","snapshot_observed_at":"2026-08-06T14:44:09.301684Z","title":"Tenenbaum, and Igor Mordatch","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.301684Z"},"links":{"cited_paper":"/paper/2305.14325","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:65733ca45f00bb32caf6c074db9d5c07f0129c8cfa52ed1974322ebbbfce68c1","observation_id":"6bb92465-f91d-41de-9fff-5c67a3152038","resolution":{"observed_at":"2026-08-06T14:44:09.301684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"gov/2010549","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.820323Z","title":null,"venue":null,"work_id":"886e5f58-f823-4c7d-b0fa-5871bd247c8c","year":2008},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.304648Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c1d37e7f030237f1779ac1b6c29dd72b69a7af3fbb9bfc9f21279beb62a1893b","observation_id":"a810d3f1-b2a7-4cbb-9524-d02ddea554e3","resolution":{"observed_at":"2026-08-06T14:44:09.824746Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04872","last_updated":"2025-12-23T02:23:47Z","snapshot_observed_at":"2026-08-04T15:54:46.196160Z","submitted_at":"2024-11-07T17:07:35Z","title":"FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04872","snapshot_observed_at":"2026-08-06T14:44:09.307985Z","title":"Frontiermath: A benchmark for evaluating advanced mathematical reasoning in ai, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.307985Z"},"links":{"cited_paper":"/paper/2411.04872","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:60694dfd4b10977a5a3412cd1a5b0bd2999c2f349db2410903f62828a1bbfebe","observation_id":"834de928-4ab6-4b41-a962-c2fad6bd8d13","resolution":{"observed_at":"2026-08-06T14:44:09.307985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.818250Z","title":"Time travel in llms: Tracing data contamination in large language models","venue":null,"work_id":"1dbaed3e-3cce-49cb-b3b8-4abac6a63ea0","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.311518Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:8b0aa0ca2d3e8d75a5787000c1e637d9a322ce8de4a9df0e6316c223e372bec8","observation_id":"0a8ff0aa-536d-4f47-a10f-ed347fddadcb","resolution":{"observed_at":"2026-08-06T14:44:11.821333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T14:44:09.314332Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.314332Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:51415b57150fae2d621e3d3be9761647673d06d11bcf037961ae8d634d7b81fd","observation_id":"29f26e3a-ff53-43ea-8d89-28c698a42493","resolution":{"observed_at":"2026-08-06T14:44:09.314332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-06T14:44:09.317974Z","title":"A survey on llm-as-a-judge","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.317974Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5622fecf4957bd666b766d311ed794d357ef35423c21ac6b3d8ea9987bdc85f2","observation_id":"3f84371c-551c-4272-9c46-4b5923fea23e","resolution":{"observed_at":"2026-08-06T14:44:09.317974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20245","last_updated":"2025-02-10T21:17:54Z","snapshot_observed_at":"2026-08-01T05:47:51.533388Z","submitted_at":"2024-10-26T18:21:44Z","title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","version":2},"cited_work":{"arxiv_id":"2410.20245","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.20245","snapshot_observed_at":"2026-08-06T14:44:09.724072Z","title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","venue":"cs.CL","work_id":"3f97366e-80a0-4e57-a83a-a721858127d0","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.321034Z"},"links":{"cited_paper":"/paper/2410.20245","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:7f62721b856970c4cb8f84159e6dfdc3548474056cc2a6c5505e3a6c0d7a360c","observation_id":"52a345a1-6f49-40e1-810e-a7640175aa05","resolution":{"observed_at":"2026-08-06T14:44:09.727362Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.809614Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":"34cb5b4b-77d3-42cf-bfd1-2bce3963f88f","year":2021},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.323991Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d17728a697cdb175149f9d4f9502b9511317aac55052214d70710d3dc964ba94","observation_id":"e85e20e6-37bf-4433-a5fb-1c42a9f3b091","resolution":{"observed_at":"2026-08-06T14:44:11.812461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.799532Z","title":"Trueskill : A bayesian skill rating system","venue":null,"work_id":"67550a25-d637-4cd0-a495-7917b4c08486","year":2006},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.327209Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d17c509794a567d7d73635c3a54e5cf5f2ce796a5a71b8f4e1acae70f0354215","observation_id":"1af5e0d3-7f8a-4b50-b586-318c67eb5b8b","resolution":{"observed_at":"2026-08-06T14:44:11.803349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.330561Z","title":"Lo RA : Low-rank adaptation of large language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.330561Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:0b2846505e3519e2e66b9cfb378d126f42e70e759217f7005b53be1908c8ef65","observation_id":"5126eeb1-cf26-4ab6-a34d-14f7327f25db","resolution":{"observed_at":"2026-08-06T14:44:09.330561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1805.00899","last_updated":"2018-10-22T17:36:07Z","snapshot_observed_at":"2026-08-02T15:33:17.783178Z","submitted_at":"2018-05-02T16:27:32Z","title":"AI safety via debate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.00899","snapshot_observed_at":"2026-08-06T14:44:09.333252Z","title":"Ai safety via debate","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.333252Z"},"links":{"cited_paper":"/paper/1805.00899","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:fe9da43f45ff9a2a3c7f3a9f0b4eb889113a28ac19d414bc26b287b01ecf8d32","observation_id":"47a8eb31-62ee-40b4-88d8-7030916e14a2","resolution":{"observed_at":"2026-08-06T14:44:09.333252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-06T14:44:09.336865Z","title":"Jiang, Alexandre Sablayrolles, Arthur Mensch, et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.336865Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:85fff108a55b1b17596397fbdf01b7d540609a7ab08f6e97ea6a9ac9d3993da0","observation_id":"c9630232-1386-4a80-9503-c007f5e98ef4","resolution":{"observed_at":"2026-08-06T14:44:09.336865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-07T13:04:50.040909Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-06T14:44:09.340177Z","title":"Jiang, Alexandre Sablayrolles, Antoine Roux, et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.340177Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e5098a4efbc98a6b4a69c4aa62846306695cbbb0c42050777e41290d893cf53c","observation_id":"b86f7a43-fbad-48c8-ab59-5f59d018b4c0","resolution":{"observed_at":"2026-08-06T14:44:09.340177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.784244Z","title":"Bowman, Tim Rockt \\\"a schel, and Ethan Perez","venue":null,"work_id":"0efda67f-918e-4c8b-a548-398f95a548fc","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.343125Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ebe59f8da9beba44153bfcb979043c2bb95d26fcf91b502f562d9b56fcbee77c","observation_id":"87e694a9-d61b-4e1a-be24-bfeeeeac2a80","resolution":{"observed_at":"2026-08-06T14:44:11.787423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13124","last_updated":"2025-01-21T05:36:13Z","snapshot_observed_at":"2026-07-06T20:24:36.194914Z","submitted_at":"2025-01-21T05:36:13Z","title":"Debate Helps Weak-to-Strong Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13124","snapshot_observed_at":"2026-08-06T14:44:09.346312Z","title":"Debate helps weak-to-strong generalization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.346312Z"},"links":{"cited_paper":"/paper/2501.13124","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:fd731c392b7ebc89c777a2289841050d645de0d141479ffa0c0cf3ddfa100876","observation_id":"325bd2a9-eb0f-4a67-a6c4-a13267cefb10","resolution":{"observed_at":"2026-08-06T14:44:09.346312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05579","last_updated":"2024-12-10T05:49:12Z","snapshot_observed_at":"2026-07-31T01:42:39.468673Z","submitted_at":"2024-12-07T08:07:24Z","title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05579","snapshot_observed_at":"2026-08-06T14:44:09.349366Z","title":"Llms-as-judges: A comprehensive survey on llm-based evaluation methods","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.349366Z"},"links":{"cited_paper":"/paper/2412.05579","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:3c8e7e14222a008d015a891b37746e37228d0bf675ddafd7b3ffc8ee3da5db09","observation_id":"dcdbfae7-ea09-477a-93c7-932389ee3b53","resolution":{"observed_at":"2026-08-06T14:44:09.349366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09212","last_updated":"2024-01-17T19:09:57Z","snapshot_observed_at":"2026-08-04T18:52:19.083847Z","submitted_at":"2023-06-15T15:49:51Z","title":"CMMLU: Measuring massive multitask language understanding in Chinese","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.09212","snapshot_observed_at":"2026-08-06T14:44:09.352447Z","title":"Cmmlu: Measuring massive multitask language understanding in chinese","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.352447Z"},"links":{"cited_paper":"/paper/2306.09212","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d7fd99df9b63b20d0c8d218c595e6b2b1ddeb2acb92bb8c7e7647cf96923ffe4","observation_id":"88498f70-5335-422b-be70-d39ac2d85e91","resolution":{"observed_at":"2026-08-06T14:44:09.352447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19485","last_updated":"2024-10-25T11:41:27Z","snapshot_observed_at":"2026-07-06T19:39:38.201130Z","submitted_at":"2024-10-25T11:41:27Z","title":"A Debate-Driven Experiment on LLM Hallucinations and Accuracy","version":1},"cited_work":{"arxiv_id":"2410.19485","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.19485","snapshot_observed_at":"2026-08-06T14:44:09.655768Z","title":"A Debate-Driven Experiment on LLM Hallucinations and Accuracy","venue":"cs.CL","work_id":"9e695b87-a12e-4f66-a373-2e852c598c4c","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.355365Z"},"links":{"cited_paper":"/paper/2410.19485","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:55a804f02be17651edd8c3c448dadcf54c6dcf1bbb8876208cc47a24fa9b8f1d","observation_id":"87452799-2260-4345-bd6e-15bd193c9117","resolution":{"observed_at":"2026-08-06T14:44:09.659401Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.774579Z","title":"Manning, Christopher R \\' e , Diana Acosta - Navas, Drew A","venue":null,"work_id":"1e3073a5-0ec7-44bd-94fd-8df3725548dc","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.358457Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:70b0375b0adac5f43b3038990f6d9ec33046e3cf54ce3ff3f3a8f2777bf11187","observation_id":"ed2720b2-bae9-47a1-b47c-1b33256a0864","resolution":{"observed_at":"2026-08-06T14:44:11.777855Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.733575Z","title":"Encouraging divergent thinking in large language models through multi-agent debate","venue":null,"work_id":"8e1ad76f-dfa8-4fd7-8f44-935fa3d1881e","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.361156Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:60d0dfb91a171400027a8f85d399356c48c9f85e1c6798edd51b42becdb09a3d","observation_id":"b16022a5-2bc7-40ea-b7db-0dbdc8a78cd6","resolution":{"observed_at":"2026-08-06T14:44:11.754017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.646538Z","title":"An empirical analysis on large language models in debate evaluation","venue":null,"work_id":"b0976f66-e2af-41df-8e3d-6c80d08c317b","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.364110Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e8658c4feb3ad8b52f14b63ccdcd961c90926e1910aeedf84ca42a55ed986cd3","observation_id":"bcd0f28c-7357-451f-a019-43dd1ffd8c48","resolution":{"observed_at":"2026-08-06T14:44:11.699063Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.595088Z","title":"The Llama 4 herd: The beginning of a new era of natively multimodal AI innovation","venue":null,"work_id":"e2d5fb32-2a65-42f5-bfb3-6111e06e6f30","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.367183Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c44d5792baae52ae813689629f02ace8ca09974782d7abcfcde53f1e279567c4","observation_id":"c70ea09d-581d-4baf-a8ff-a4403bcf8c35","resolution":{"observed_at":"2026-08-06T14:44:11.638130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.341","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.546852Z","title":"Discursive socratic questioning: Evaluating the faithfulness of language models' understanding of discourse relations","venue":null,"work_id":"7f7f433d-4bae-426c-b772-71a143ceec84","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.370593Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:1b8e3df0e83d3c5ff1479f8c0d69ed28ce971b7d1cec250cb381f1021cca802b","observation_id":"658269b3-bea3-41b0-9ee6-c5226b858a26","resolution":{"observed_at":"2026-08-06T14:44:09.550141Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-06T14:44:09.373495Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs , 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.373495Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:a3765ddc9528c1ea18ae40fb5b3ac68aea3c4d5738df6c0673fab3323023d8cf","observation_id":"963a0a2e-6217-4657-988a-4b2724e01846","resolution":{"observed_at":"2026-08-06T14:44:09.373495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.336032Z","title":"Cheaper, better, faster, stronger","venue":null,"work_id":"c151b123-5f9b-4931-b203-da40d4fccb7d","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.376438Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:4d55622b98afcf74378b18fe27e32e748a3ad9702927b98dab809a19050c86c5","observation_id":"471f0037-bffb-4f85-927f-10aaadc82ad1","resolution":{"observed_at":"2026-08-06T14:44:11.454555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:11.202204Z","title":"Mistral large","venue":null,"work_id":"505b3ec1-455e-4ca2-9130-2ad2069aac4a","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.379133Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:d5e0d45bda5bc137464c2bd64d871f6186a012d2168fb4a22462ecbceb190c21","observation_id":"48907f32-9cc1-47f7-a2bb-4fc3af06a2d2","resolution":{"observed_at":"2026-08-06T14:44:11.325353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11044","last_updated":"2025-02-07T21:56:40Z","snapshot_observed_at":"2026-07-06T18:31:49.437336Z","submitted_at":"2024-06-16T19:02:31Z","title":"Evaluating the Performance of Large Language Models via Debates","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11044","snapshot_observed_at":"2026-08-06T14:44:09.381842Z","title":"Evaluating the performance of large language models via debates","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.381842Z"},"links":{"cited_paper":"/paper/2406.11044","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ca71cb96ff4d8eb0554f392c62f460d3546df2b9a87b2e4998c8c80f1ba1d4e8","observation_id":"e966dac2-cd50-4d15-baaa-94f3454d0b18","resolution":{"observed_at":"2026-08-06T14:44:09.381842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T14:44:09.384917Z","title":"Gpt-4 technical report, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.384917Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c21c85ac6de63bcf21ad83b2e2c432964fc74674bd69e228c620597340ee81ad","observation_id":"0933da57-b03d-4223-9983-e35edb8fc8f0","resolution":{"observed_at":"2026-08-06T14:44:09.384917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-06T14:44:09.387932Z","title":"GPT-4o System Card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.387932Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:f524eb737c2fd1d6a42927dcdf8e4b60b9258c2c9fcf105cc715f9b57caa40a3","observation_id":"ec0c7b9f-1c02-404d-b2f5-9f6bb2b5342d","resolution":{"observed_at":"2026-08-06T14:44:09.387932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.958064Z","title":"GPT-4o mini: advancing cost-efficient intelligence","venue":null,"work_id":"f921298a-6533-44b8-8fb2-cdbfaaf6d1da","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.391394Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ed905deda057bab2c4e4d458ce5244f47c477c95d3cb262ce17ced314a0f0bec","observation_id":"f817549b-8f37-4ac2-800d-c183768200b8","resolution":{"observed_at":"2026-08-06T14:44:11.063257Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-06T14:44:09.394084Z","title":"Openai o1 system card, 2024 c","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.394084Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:6a16398d0b18ef5ae3cf55bc5413bc44e33860e5ba300e384c94d91e541d8059","observation_id":"8c4d6f4f-37f0-4b11-a6fc-4272bd79a562","resolution":{"observed_at":"2026-08-06T14:44:09.394084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.396998Z","title":"Chatterji, Faisal Ladhak, and Tatsunori Hashimoto","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.396998Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:c182b400750066e54fb2b0fd24ca74e840b21699542fed8b3a2290407de0dcf3","observation_id":"81b8e3f6-eb82-4a21-94f3-71c9f31dfe3d","resolution":{"observed_at":"2026-08-06T14:44:09.396998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.399824Z","title":"Mapping global dynamics of benchmark creation and saturation in artificial intelligence","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.399824Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:3b101b2ff0dd12a6eb0e13dc2c9c8fb82d6d51a6a4dc971040b1341cb630bab1","observation_id":"067938fe-4300-44b8-b2e4-4beba9569ad5","resolution":{"observed_at":"2026-08-06T14:44:09.399824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-06T14:44:09.402625Z","title":"Humanity's last exam","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.402625Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:df572e60eeacd7d95977ec1c89273eac4f2942bee0fc02d556028480ff3390df","observation_id":"7053c87d-d62c-44aa-b800-fd743facd11e","resolution":{"observed_at":"2026-08-06T14:44:09.402625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.699050Z","title":"Introducing gemini 2.0: our new ai model for the agentic era","venue":null,"work_id":"46745679-bebd-4496-9a14-2773208d0950","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.405513Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:015599964f4bcfc4a89b91e3bab0df8811611718833a3a9f2a46238a5db43ac8","observation_id":"f12d1973-fa38-4f70-9902-f1b6e486e20c","resolution":{"observed_at":"2026-08-06T14:44:10.854039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.359908Z","title":"Multi-layered evaluation using a fusion of metrics and LLMs as judges in open-domain question answering","venue":null,"work_id":"fc5290ca-3f1b-433b-ab46-56b2badac6f4","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.408457Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:2ff30fa193e4d6ad7bbcc10b335b5b3e17f7e782c95e0feda3c03e71d6cad99e","observation_id":"36bb7fd3-ed87-4e09-b768-ccd3d1c8dc91","resolution":{"observed_at":"2026-08-06T14:44:10.471219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-04T22:55:15.345443Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-06T14:44:09.411020Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.411020Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:afca59f902fbbb8f3c83f7fffa60a976d36a2a0ab82f5b2203dba400af578a1e","observation_id":"07618cbf-e2a8-490c-bec3-fbc58bbe8bf3","resolution":{"observed_at":"2026-08-06T14:44:09.411020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.413880Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.413880Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ba1b5cbbd3919b973fe6f35ab74244f33d4536339d29c1db3b0bbd10c9fbef11","observation_id":"8bb4c05e-68cc-4951-855a-2f612619ae98","resolution":{"observed_at":"2026-08-06T14:44:09.413880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08632","last_updated":"2023-09-13T19:47:33Z","snapshot_observed_at":"2026-07-06T16:19:10.103540Z","submitted_at":"2023-09-13T19:47:33Z","title":"Pretraining on the Test Set Is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08632","snapshot_observed_at":"2026-08-06T14:44:09.416573Z","title":"Pretraining on the test set is all you need, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.416573Z"},"links":{"cited_paper":"/paper/2309.08632","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:1714e8a642400bd2c489d60c3a343e5160148be4a597ebf24cc803b04b98b2d4","observation_id":"3cf9e6df-393b-4b02-adde-6800a7912adf","resolution":{"observed_at":"2026-08-06T14:44:09.416573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.294425Z","title":"Detecting pretraining data from large language models","venue":null,"work_id":"ed45d7cd-f59d-4609-8510-d3d53eab2bb0","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.419654Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:6f189bcc8d22d8a896ff2967797ab241cd29b58b2a674dfa81e5145824b57af6","observation_id":"d5675478-26e6-415f-b752-bc3c18d2885d","resolution":{"observed_at":"2026-08-06T14:44:10.319102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.247811Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models","venue":null,"work_id":"a00addbd-bb3e-4144-9b0e-d163b3ba8651","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.422236Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:222c3013c2d856465a5431edc0119d45cc7a04827834c2f04c7deaeaa1b9c07b","observation_id":"db9419df-7567-4932-9c0b-f3dd29fabb58","resolution":{"observed_at":"2026-08-06T14:44:10.271308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02257","last_updated":"2024-10-15T18:37:03Z","snapshot_observed_at":"2026-07-06T19:10:04.150678Z","submitted_at":"2024-09-03T19:31:03Z","title":"MMLU-Pro+: Evaluating Higher-Order Reasoning and Shortcut Learning in LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02257","snapshot_observed_at":"2026-08-06T14:44:09.424893Z","title":"MMLU-Pro+: Evaluating Higher-Order Reasoning and Shortcut Learning in LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.424893Z"},"links":{"cited_paper":"/paper/2409.02257","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:40c546fb2bbbc24da08d5d262096626ba70e84068112f2abe497c82ff36c3591","observation_id":"9717d2a1-315a-4ac1-b51d-f7940e0fb2e5","resolution":{"observed_at":"2026-08-06T14:44:09.424893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.180277Z","title":null,"venue":null,"work_id":"3445b39c-4ef7-44be-bc45-c5274d3f8bdb","year":2019},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.427865Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:a89b2e20b36aec1532ebc6e6a255c7dbfd3e0e33c83c483aa1b7fa31d707435d","observation_id":"1647dcfc-81fe-4705-a412-fd81c0ab1680","resolution":{"observed_at":"2026-08-06T14:44:10.208592Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.138234Z","title":null,"venue":null,"work_id":"cb6f3d19-ec92-4189-bd17-faa965bd78fe","year":2019},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.430445Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:504fdbbddb318f7bf430bffc7c508e69112108a95970b45f506cc8fb3648a7a6","observation_id":"e7e1d3ac-fd53-430f-8ac1-ee4cd9cb21f5","resolution":{"observed_at":"2026-08-06T14:44:10.148317Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.433045Z","title":"Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.433045Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e03bbe0e08ff411e453b686ee7fc70d71247b690cf6edd7b29d244f7459cc36d","observation_id":"29e7c9eb-c1a0-4b40-a774-be1a12be7040","resolution":{"observed_at":"2026-08-06T14:44:09.433045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.121451Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","venue":null,"work_id":"5bfdf7da-abe0-4332-9d29-b567c191c161","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.436267Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:e420213c49344aa3f6426794573b88b187f7073f70aea92a5126739d4be81223","observation_id":"afab2f8b-6539-46ef-83d0-73bada5393a3","resolution":{"observed_at":"2026-08-06T14:44:10.124690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.111868Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"e5209b90-1834-49a7-9e24-1dff850b2b9a","year":2022},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.439020Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:2dbcedd02b563bb289a80f63e18944d5c032a3ed2be53a199d77403051350441","observation_id":"ff3d821e-b5f0-47d2-aae8-11173c64161c","resolution":{"observed_at":"2026-08-06T14:44:10.115093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.102582Z","title":"Livebench: A challenging, contamination-free LLM benchmark","venue":null,"work_id":"39c61bca-80e8-410f-83de-2b2d03ea8ac7","year":2025},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.441678Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:5b86c5457a477731b451fb77452a9d089e0a427bef55de5bf2877aadcb61ca20","observation_id":"6c16eb36-7f0b-4507-a872-929d4fc131ee","resolution":{"observed_at":"2026-08-06T14:44:10.105651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.emnlp-main.325","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.500027Z","title":"QUD eval: The evaluation of questions under discussion discourse parsing","venue":null,"work_id":"3de6c37e-5c39-43c7-8ff5-6daae025714e","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.444655Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ad9be9296bccb6a773aa395b8a05faedd137c77d3658ef47da8f761d54d5cb95","observation_id":"ff34c8d0-bf09-4d70-9d71-6d5258bfea53","resolution":{"observed_at":"2026-08-06T14:44:09.505113Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-06T14:44:09.448188Z","title":"Benchmark data contamination of large language models: A survey, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.448188Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:50fc8974af6d551c3a077372b1bd39eb3a25305a50139f30ea940b8e82293dac","observation_id":"f51edf2e-e3ea-4b73-b8f5-05d222f22c20","resolution":{"observed_at":"2026-08-06T14:44:09.448188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15043","last_updated":"2024-06-03T06:02:39Z","snapshot_observed_at":"2026-07-06T17:34:20.247808Z","submitted_at":"2024-02-23T01:30:39Z","title":"KIEval: A Knowledge-grounded Interactive Evaluation Framework for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15043","snapshot_observed_at":"2026-08-06T14:44:09.451179Z","title":"Kieval: A knowledge-grounded interactive evaluation framework for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.451179Z"},"links":{"cited_paper":"/paper/2402.15043","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:deaadbe599ec94ee6ccc9b27d8a565d2fa48c2510c6fb1fa94322aa70510188d","observation_id":"faaed509-9aa0-41be-8f71-1d8c24b38fac","resolution":{"observed_at":"2026-08-06T14:44:09.451179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.20267","last_updated":"2024-10-07T02:53:44Z","snapshot_observed_at":"2026-08-05T16:46:54.515628Z","submitted_at":"2024-05-30T17:19:19Z","title":"Auto-Arena: Automating LLM Evaluations with Agent Peer Battles and Committee Discussions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.20267","snapshot_observed_at":"2026-08-06T14:44:09.454188Z","title":"Auto-arena: Automating llm evaluations with agent peer battles and committee discussions, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.454188Z"},"links":{"cited_paper":"/paper/2405.20267","citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:995ae815fb9a8b7a33bbc8874921606636669c84c4c4d3e83e76c935b1b58f02","observation_id":"d1754a25-0780-4bf6-8e85-c1d0fa4eb958","resolution":{"observed_at":"2026-08-06T14:44:09.454188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.092478Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":"d3a9bf1b-79c3-4821-b941-488bf0324094","year":2023},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.457014Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:eea333635bf3309f5acad4ca47720a169465a14f899a93b62f3d6f60f9639de7","observation_id":"28e1a61d-5759-49a8-a35d-c52eb1b52a35","resolution":{"observed_at":"2026-08-06T14:44:10.096336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:10.082831Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"6b7b406a-8385-435b-8c26-122c3cb48547","year":2024},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.459891Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:11e357adf9d46a7cc088e47c5a9237ab482797f6845d1e0de793c8c09aac532b","observation_id":"aa7be3b1-c45d-452b-a424-1fc0ded6b452","resolution":{"observed_at":"2026-08-06T14:44:10.086077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.462617Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.462617Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ed3640440ee1b48d3d3addab2f2a99373939944935ea750079e7b559eb229629","observation_id":"7d0e033a-3e88-47c9-bea6-db99e9f6c28f","resolution":{"observed_at":"2026-08-06T14:44:09.462617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.465929Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.465929Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:b784b944adef4003e7760643b112d6cd437b421b3d81bdaafcca970a09c2fd55","observation_id":"6edf6c22-fd91-48f8-8d29-35f64be39ffd","resolution":{"observed_at":"2026-08-06T14:44:09.465929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.468994Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.468994Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:1c3aac01433218b5e6c60f9205a384a78a7b698253e3a4b051d08915d97bc81a","observation_id":"4339ae32-1247-4087-955b-68eaa972ae45","resolution":{"observed_at":"2026-08-06T14:44:09.468994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:44:09.471855Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-06T14:44:09.471855Z"},"links":{"citing_paper":"/paper/2507.17747"},"observation_digest":"sha256:ddc5345093e68ae4837c141227e3f666da3ea17cc30c3c655455e01342b8520d","observation_id":"5fde41a7-2a87-47c6-9831-2c3a0ca7d3a2","resolution":{"observed_at":"2026-08-06T14:44:09.471855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.17747","last_updated":"2025-08-08T01:56:30Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T14:37:08.284816Z","submitted_at":"2025-07-23T17:58:14Z","title":"Pretraining on the Test Set Is No Longer All You Need: A Debate-Driven Approach to QA Benchmarks"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":44,"verified_exact":5,"verified_fuzzy":30},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 0 inbound Pith citation observations for arXiv:2507.17747."}