{"as_of":"2026-08-21T08:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fc01539775f4617a6ea152b892bfc4b6df1e913c134d3899a377ca1c9ed39905","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T18:01:55.836203Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T05:41:33.026345Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-12T05:46:24.255634Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"cited_work":{"arxiv_id":"2507.19219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.19219","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How much do large language models cheat on evaluation? benchmarking overestimation under the one-time-pad-based framework","venue":null,"work_id":"318643a3-d703-42d2-b782-651ae7d6b561","year":2025},"citing_paper":{"arxiv_id":"2605.02442","last_updated":"2026-05-04T10:42:26Z","snapshot_observed_at":"2026-08-15T23:26:09.618748Z","submitted_at":"2026-05-04T10:42:26Z","title":"Measuring AI Reasoning: A Guide for Researchers","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-08T18:53:18.586923Z"},"links":{"cited_paper":"/paper/2507.19219","citing_paper":"/paper/2605.02442"},"observation_digest":"sha256:ee1985eb499d953eb1c1b8d236a444c93a7ebfb2de1c6026bc361de5f4b1d58b","observation_id":"ebece4ca-6d1a-44e5-a104-5cf02be18e75","resolution":{"observed_at":"2026-05-26T02:03:01.207018Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"cited_work":{"arxiv_id":"2507.19219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.19219","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How much do large language models cheat on evaluation? benchmarking overestimation under the one-time-pad-based framework","venue":null,"work_id":"318643a3-d703-42d2-b782-651ae7d6b561","year":2025},"citing_paper":{"arxiv_id":"2605.10810","last_updated":"2026-05-15T15:01:38Z","snapshot_observed_at":"2026-08-21T03:37:15.843744Z","submitted_at":"2026-05-11T16:32:06Z","title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T05:41:33.026345Z"},"links":{"cited_paper":"/paper/2507.19219","citing_paper":"/paper/2605.10810"},"observation_digest":"sha256:c2e201fad87fefbe81e07b019fc383a065f06ecfe69c039d49e7844719e948cf","observation_id":"6a1e66ce-ddef-4ac5-a4d7-52e11b478d8e","resolution":{"observed_at":"2026-05-26T02:03:01.207018Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.19219/citation-record","integrity":"/paper/2507.19219/integrity","json":"/paper/2507.19219/citation-record.json","paper":"/paper/2507.19219"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.508876Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.508876Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:a73608eff42d2ee42bbcd1c69730495c63f9a7089fb7796f5412998db3ea2a9e","observation_id":"de701f72-c710-4c19-aa79-1cdb149ba1a5","resolution":{"observed_at":"2026-08-15T18:01:55.508876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.518400Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.518400Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:f5f92649188fc20e22ca4db27b24033af4d9b01771e128b06149bf6ca46c6644","observation_id":"ef661ef1-c1da-4461-bfe7-e42144758803","resolution":{"observed_at":"2026-08-15T18:01:55.518400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.526142Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.526142Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:1c746d1541ab2f4630e19217179d98b95746e73ad790947d6f1a6c6278792f75","observation_id":"a23bedcb-7cda-496c-9e55-d50cec278a01","resolution":{"observed_at":"2026-08-15T18:01:55.526142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.533594Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.533594Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:3d7a9d65140086baf9f55c11db23bf7d8c1c0f22714cec92fb9ffb193919be81","observation_id":"25c9e4e2-3691-4a75-bddf-066fb4661db1","resolution":{"observed_at":"2026-08-15T18:01:55.533594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.844339Z","title":"Qwen2 Technical Report","venue":null,"work_id":"5467a271-b2f1-42fb-95be-21eed1a9773a","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.541887Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:aaafbde17895c24d9262f9c178716eb30d5a869ef74ef8f964b542a3ba933ea2","observation_id":"d3a171cd-56dd-4186-95aa-4a526f425728","resolution":{"observed_at":"2026-08-15T18:01:56.851023Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.820741Z","title":"G.; and Chapelle, C","venue":null,"work_id":"32345a5c-d276-43fe-a043-6e440d7f11b8","year":1992},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.552138Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:8e81bb1865845b5d987e70e906c2ae93032fc8bd03bb721f7e01f3e618b062fc","observation_id":"d1b9016e-2025-468d-aa54-c3a51a77b0aa","resolution":{"observed_at":"2026-08-15T18:01:56.828269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.559260Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.559260Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:4a1b09a28afd28f9fb292307224f82cacc0fb5ba26b736f77ef7c644105a36bd","observation_id":"bf7bd0dc-f3de-4581-8a44-4ccb29f39f76","resolution":{"observed_at":"2026-08-15T18:01:55.559260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.787415Z","title":null,"venue":null,"work_id":"131c45b5-49d0-481c-9079-b4797a884ab1","year":1979},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.566010Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:1b99afcbd8accc5c79e914ce5c3241f0c18998669988c62308ac6929d7bbbd81","observation_id":"20545eb8-2c45-488e-88b6-b1014d396d38","resolution":{"observed_at":"2026-08-15T18:01:56.793642Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.767321Z","title":null,"venue":null,"work_id":"f1ed2b74-5d51-434a-ba4e-856df98039c6","year":1968},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.573762Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:08eea0b45c46a4836c8cb91ce5f446bf6feafb2246855bd0988d26f85b3b4bb8","observation_id":"1a6acc29-5978-4d64-a9fa-a3e88cf3e8f3","resolution":{"observed_at":"2026-08-15T18:01:56.772998Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.748155Z","title":null,"venue":null,"work_id":"bd742c6c-1205-411d-807a-5f41336ebd4b","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.582581Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:211d73782c2c8a033843f1571a792f8e579437268d09541c8427edd7103314e1","observation_id":"909831b6-c8c3-4b24-bc45-54b8228231d0","resolution":{"observed_at":"2026-08-15T18:01:56.754109Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.728192Z","title":"N.; Li, T.; Li, D.; Zhu, B.; Zhang, H.; Jordan, M","venue":null,"work_id":"ced8419b-bce8-4d72-8c94-04b41c4c7838","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.592757Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:b85c0ec1f4f87243b88804e0302676c1beb13286d774be2a3b63f900bde2478e","observation_id":"a6f395b1-a8f4-467d-9e14-e60b4c8dceab","resolution":{"observed_at":"2026-08-15T18:01:56.734151Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-15T18:01:55.602383Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.602383Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:7807729671922f43bb123ae36c5ecbf600761a5a806d743934d541a02cdd45be","observation_id":"975ef086-0d2f-4479-be7e-f50b7669ff4f","resolution":{"observed_at":"2026-08-15T18:01:55.602383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.708396Z","title":null,"venue":null,"work_id":"b22e038b-f6a5-4665-9123-baa8604fe0ed","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.608638Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:bd96a3b292c6024964d49aa4e55bf84290fdbf9b53556193d1a50c4a34c284fb","observation_id":"c2580364-1796-4722-9084-5472b51f73ac","resolution":{"observed_at":"2026-08-15T18:01:56.714746Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.688477Z","title":null,"venue":null,"work_id":"d81129b0-158e-4ef6-98df-630c109a6d1d","year":1967},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.614316Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:cb2ea31d755b4676afe047cdc3efb59bd5d4f75bc72fb942ecb45997cf98bdec","observation_id":"38f91d24-a655-4e12-bc39-52964c09dd74","resolution":{"observed_at":"2026-08-15T18:01:56.694654Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.619561Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.619561Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:69c08f7963b7f4063db822d4499603b5b2c47b565d3bb6631443ca0aa875fbf0","observation_id":"5c996d1c-ca6c-4adf-b9bd-3b7c14346336","resolution":{"observed_at":"2026-08-15T18:01:55.619561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11644","last_updated":"2023-10-02T06:12:30Z","snapshot_observed_at":"2026-08-13T11:19:55.436754Z","submitted_at":"2023-06-20T16:14:25Z","title":"Textbooks Are All You Need","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11644","snapshot_observed_at":"2026-08-15T18:01:55.624898Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.624898Z"},"links":{"cited_paper":"/paper/2306.11644","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:249d9d9ad9a5215fdc50f4681aae3da35aa78382dfd28d88c64ccb5f20fe7f7e","observation_id":"197b8bb9-3113-42e5-838b-060263718a05","resolution":{"observed_at":"2026-08-15T18:01:55.624898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.630645Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.630645Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:5154cb3733a6806715da0046d916a4f655ed6d4021be46949560e690a79c9fd2","observation_id":"adee98e0-fa9c-4f64-952b-435a2bd005f3","resolution":{"observed_at":"2026-08-15T18:01:55.630645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12753","last_updated":"2025-03-06T12:55:25Z","snapshot_observed_at":"2026-08-16T13:41:46.107232Z","submitted_at":"2024-06-18T16:20:53Z","title":"OlympicArena: Benchmarking Multi-discipline Cognitive Reasoning for Superintelligent AI","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12753","snapshot_observed_at":"2026-08-15T18:01:55.636266Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.636266Z"},"links":{"cited_paper":"/paper/2406.12753","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:dd18769fadfb61e57df33c9618c8d9162cd0c31a8ed1dd7583c09fd507569d1a","observation_id":"64af435e-867f-448a-94c5-e241839bdff4","resolution":{"observed_at":"2026-08-15T18:01:55.636266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.642060Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.642060Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:edbad30f645f3404600a7a505854133e0260b47edfe87c606893e9c8ddeefb4d","observation_id":"d6005ea8-7433-46bc-bc2d-d750f52730c1","resolution":{"observed_at":"2026-08-15T18:01:55.642060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06059","last_updated":"2024-01-11T17:24:49Z","snapshot_observed_at":"2026-08-17T18:53:54.955991Z","submitted_at":"2024-01-11T17:24:49Z","title":"Investigating Data Contamination for Pre-training Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06059","snapshot_observed_at":"2026-08-15T18:01:55.648414Z","title":"Z.; Zhong, M.; Schaeffer, R.; Ouyang, S.; Han, J.; and Koyejo, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.648414Z"},"links":{"cited_paper":"/paper/2401.06059","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:73e8498d4983f7d490c737878d2d92ea65930149bd42c35e7de5c73b7bc60c41","observation_id":"f8adae2e-d0ec-4b1d-9a8d-fc7ace04c78c","resolution":{"observed_at":"2026-08-15T18:01:55.648414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.624869Z","title":"E.; Yang, J.; Wettig, A.; Yao, S.; Pei, K.; Press, O.; and Narasimhan, K","venue":null,"work_id":"c38061a9-0d65-4399-b330-80191b2fbc75","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.654260Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:e3fa4a9506a0524e0e6aa04d2c18d5f7a7ab48ea6e5bb33fd2096690ef20840f","observation_id":"9793d50a-88dd-449c-9dcc-7cda7679c799","resolution":{"observed_at":"2026-08-15T18:01:56.631862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11939","last_updated":"2024-10-14T18:11:58Z","snapshot_observed_at":"2026-08-21T06:07:47.921030Z","submitted_at":"2024-06-17T17:26:10Z","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11939","snapshot_observed_at":"2026-08-15T18:01:55.662468Z","title":"E.; and Stoica, I","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.662468Z"},"links":{"cited_paper":"/paper/2406.11939","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:f03d2c91a9c15cd9c8eba00939ace556a90c09f47cf9c895ce5eb535eae38a38","observation_id":"75c7a7a7-ee09-46b9-993b-f3fd2f476ff1","resolution":{"observed_at":"2026-08-15T18:01:55.662468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.603513Z","title":"E.; and Stoica, I","venue":null,"work_id":"bfdf9aaa-d4e5-4aa9-9062-55afaa7090de","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.668498Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:109a6ddca8dd92a439a7ec8de92ccaac5e3cc0b961c1ae2e0f56c8d5cd25afa9","observation_id":"32f46696-e39d-45a0-8542-9ef7e0fbd0fb","resolution":{"observed_at":"2026-08-15T18:01:56.610206Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.05463","last_updated":"2023-09-11T14:01:45Z","snapshot_observed_at":"2026-08-02T22:47:03.212781Z","submitted_at":"2023-09-11T14:01:45Z","title":"Textbooks Are All You Need II: phi-1.5 technical report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.05463","snapshot_observed_at":"2026-08-15T18:01:55.674430Z","title":"D.; Gunasekar, S.; and Lee, Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.674430Z"},"links":{"cited_paper":"/paper/2309.05463","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:1a26955af2d94e7c05e0f2230b9c12fd258d853f729ba1b35b49663aa31e379f","observation_id":"1bab9562-35e8-488f-ab1a-41570e4bb1c3","resolution":{"observed_at":"2026-08-15T18:01:55.674430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.585629Z","title":null,"venue":null,"work_id":"121733cd-5e0d-40ff-a8b5-49b55501ffb3","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.681134Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:13cd432929f0fe717327faf72efb450d7958d8b70902c91ea3282fd1e42f3778","observation_id":"50b439a7-defb-4651-8ee1-477d7e163daa","resolution":{"observed_at":"2026-08-15T18:01:56.591048Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.567479Z","title":null,"venue":null,"work_id":"c22de35b-42f7-4f76-86be-fa9b9f7cc6a9","year":2006},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.686898Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:d8f1edbaf75c6bc46d4466e1dc2f9e91cd7efc01afa38c17bde05b9b0f457c63","observation_id":"0ba9996c-c450-4e83-8b35-e85b6cf4e6e7","resolution":{"observed_at":"2026-08-15T18:01:56.573262Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-17T03:25:04.404839Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-15T18:01:55.693029Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.693029Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:a3feb4706992bcef075a9078587066aed6676a912cd68eaa1a1dc4cfb91e3a7e","observation_id":"9a1eb895-66d7-4f7d-b13a-e5a0135f118f","resolution":{"observed_at":"2026-08-15T18:01:55.693029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.545471Z","title":null,"venue":null,"work_id":"9eac2b65-3d79-4486-b614-082085b6d710","year":null},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.699070Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:639f4712c2439bf5cd9fdf31caced80b1f21466cfc0e39241857d676177f2279","observation_id":"8de1b18e-c067-44e0-9b9e-8c0f82b02c4b","resolution":{"observed_at":"2026-08-15T18:01:56.551689Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05229","last_updated":"2025-08-27T16:24:39Z","snapshot_observed_at":"2026-08-16T08:54:56.543625Z","submitted_at":"2024-10-07T17:36:37Z","title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05229","snapshot_observed_at":"2026-08-15T18:01:55.704411Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.704411Z"},"links":{"cited_paper":"/paper/2410.05229","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:3e362b3b9da9cc01f62e24d1be819d4d0307e2ac31a6245bd4448bd2888afbd9","observation_id":"10941fed-2f6d-4968-be0a-6c7fedbc8ae0","resolution":{"observed_at":"2026-08-15T18:01:55.704411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T18:01:55.709827Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.709827Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:b3bd8bd955dc84f2d5486a145132f2695a4a204bf343b2d4162e21bb17ac5206","observation_id":"a6a548f3-fcfc-425b-a430-c3c840c0c252","resolution":{"observed_at":"2026-08-15T18:01:55.709827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-15T18:01:55.715343Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.715343Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:31a51282307090db60d7ef905f8afd892661267ad9e53013365b7ca80be6efa0","observation_id":"9f01e147-ff93-4e73-9c78-df41f77cf51c","resolution":{"observed_at":"2026-08-15T18:01:55.715343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.527076Z","title":null,"venue":null,"work_id":"6b16b9cc-c434-46fa-b620-ed349bec9d6e","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.721411Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:6b7062fc461a2aa00857b7cfdd4640fd45afc8f873a8303e005f194b2822fea1","observation_id":"cb003761-1184-4789-b462-f56a05221eea","resolution":{"observed_at":"2026-08-15T18:01:56.532668Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.508734Z","title":null,"venue":null,"work_id":"e3f63934-6be8-4a6f-8162-2a20d7fbb805","year":1949},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.726781Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:2230ca894139503f20afc694a8c4ecbb9940b3ead388895d1e24910bc0d964e6","observation_id":"4204b1cb-f08f-4bb8-933d-288413222b3d","resolution":{"observed_at":"2026-08-15T18:01:56.513939Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.489290Z","title":null,"venue":null,"work_id":"565577cf-30e2-4a75-bc71-5af7b72fef56","year":2025},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.732456Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:9c3459ff9f0902f011dcf0d29d0abf3bf95f340ffa35bc692158a219106ad45f","observation_id":"886fe9bf-27e8-4f60-b7e1-4ce63b4081be","resolution":{"observed_at":"2026-08-15T18:01:56.495555Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-15T18:01:55.738519Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.738519Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:7b14027071413f66eff097afb7d16ebb4a9dab1d9b4d32e7b1d95326d75340d1","observation_id":"df5bf444-4711-4382-9608-1659f3dfa045","resolution":{"observed_at":"2026-08-15T18:01:55.738519Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.468436Z","title":null,"venue":null,"work_id":"2d4a419a-4121-4690-8aa7-b5627c44021a","year":2021},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.744656Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:aaa2923de8831889666de578f3db30f183cb6ef6e9e5b8f451668df996e3c862","observation_id":"1461311b-bddd-401c-a678-87a7554ef371","resolution":{"observed_at":"2026-08-15T18:01:56.473936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.444756Z","title":"R.; Zhang, S.; Sun, Y.; and Wang, W","venue":null,"work_id":"87b1ce73-1d3a-483b-b751-f03996a0b98b","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.750766Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:2741962abceed142659e70cba80dbeec9a4b6249dd858754acd047360fabcc5f","observation_id":"b0bb6115-a0dd-48cb-b99d-ab1f8521afe6","resolution":{"observed_at":"2026-08-15T18:01:56.451035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-20T16:26:42.002863Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-15T18:01:55.758217Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.758217Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:433502fb7c948522e4a08248627f15c268012b9b877d72427ce478b0bdb977eb","observation_id":"6ab5017e-a309-42cc-8af4-4e8d3216c147","resolution":{"observed_at":"2026-08-15T18:01:55.758217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01257","last_updated":"2025-03-06T12:13:14Z","snapshot_observed_at":"2026-08-18T05:01:06.444745Z","submitted_at":"2024-10-02T06:05:52Z","title":"HelpSteer2-Preference: Complementing Ratings with Preferences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01257","snapshot_observed_at":"2026-08-15T18:01:55.764894Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.764894Z"},"links":{"cited_paper":"/paper/2410.01257","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:9d0f8b3dcff66eb88a877ed66dea49e509fde389863d29520f6f0a41adcefcf9","observation_id":"3db0d18a-3759-437d-8980-54c6b335f7b5","resolution":{"observed_at":"2026-08-15T18:01:55.764894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-15T18:01:55.772078Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.772078Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:027d9dfbc11309189878e450a2f9f118d89ade80742806ee4a29b0bf3e1e45f0","observation_id":"5b274764-4af1-479b-8f8c-d41e653f5620","resolution":{"observed_at":"2026-08-15T18:01:55.772078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:55.779492Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.779492Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:ee83f525ff123457f32414838ac73ea51f713da13c8f7a126a4f0ac1ae363526","observation_id":"28de5af8-b5f2-4ac0-9dc9-7265f8dac8ad","resolution":{"observed_at":"2026-08-15T18:01:55.779492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.422519Z","title":null,"venue":null,"work_id":"742bdd8a-772c-4dd1-8b8a-ff3faacc5f67","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.786651Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:23a0a536531d19914bd0a029e816df9d68f27165c42a55f321d9e0024ce54380","observation_id":"1d6b58b7-c008-4b48-8b5f-6e16d748d806","resolution":{"observed_at":"2026-08-15T18:01:56.428925Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-08-17T14:15:23.870937Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-15T18:01:55.792774Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.792774Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:3d00c4d318b7d48141ab3de410dcc085a61b26c3f981a0b6d1a18b268d0a1c11","observation_id":"6f6084a4-9800-4ef4-aab1-8a4103bece7a","resolution":{"observed_at":"2026-08-15T18:01:55.792774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-16T14:44:53.286559Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-15T18:01:55.799239Z","title":"E.; and Stoica, I","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.799239Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:c49eb2cb526c97ae6d8d0feed853c7889a31218f8e7f04c9c9c2c4f636e1ce85","observation_id":"ea4ec8ab-bfca-45cd-acfa-51f14490acd4","resolution":{"observed_at":"2026-08-15T18:01:55.799239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.20311","last_updated":"2024-07-29T17:52:40Z","snapshot_observed_at":"2026-08-20T19:03:38.607906Z","submitted_at":"2024-07-29T17:52:40Z","title":"Physics of Language Models: Part 2.1, Grade-School Math and the Hidden Reasoning Process","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.20311","snapshot_observed_at":"2026-08-15T18:01:55.808143Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.808143Z"},"links":{"cited_paper":"/paper/2407.20311","citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:5d2ec2437e7b971cded4b109dc4663f36954aae42f489aeb134a5eaad57f5d41","observation_id":"be476101-66eb-474d-bf41-3bee196f84c9","resolution":{"observed_at":"2026-08-15T18:01:55.808143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.404245Z","title":null,"venue":null,"work_id":"63b1205f-0e56-44a6-af48-14cf806233a1","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.813918Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:cf0df606d3d1874a22acbb7762a7db8e2dddf264b11905a9e913bbd5e1ddb113","observation_id":"2c8fd61c-cbd6-49f4-921d-6eccb7cdbad9","resolution":{"observed_at":"2026-08-15T18:01:56.409971Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.385578Z","title":null,"venue":null,"work_id":"6055f012-6f26-4c81-a743-83bdec123800","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.820500Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:8e7136c4b59b995f1c02c3dd92a3eae3c77a1ad80a8dfc7a9caa7b0e1aa21ff9","observation_id":"f1709b14-4be9-4403-b2fa-bcbd841b2cd5","resolution":{"observed_at":"2026-08-15T18:01:56.391357Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.365749Z","title":"Z.; Yang, D.; and Xie, X","venue":null,"work_id":"01eeacdb-bcc8-4d13-9c09-9811623475c5","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.827990Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:536c0e6f52db5a39318b475e693535056ce46c6cf187522a7ca08f81b9c36749","observation_id":"91f5e9fc-20ee-4e8d-ac92-aef7568cd88e","resolution":{"observed_at":"2026-08-15T18:01:56.371048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T18:01:56.342428Z","title":null,"venue":null,"work_id":"01fc9738-2857-460d-b2a8-075f2926c616","year":2024},"citing_paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-15T18:01:55.836203Z"},"links":{"citing_paper":"/paper/2507.19219"},"observation_digest":"sha256:f5fb744fe78712c9af3617fa14243beef1fd91ff643321e3f4cdb0d1dee5adb0","observation_id":"ac12e6b6-1d26-47f5-9171-0072111cbdab","resolution":{"observed_at":"2026-08-15T18:01:56.351817Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.19219","last_updated":"2026-05-24T04:42:35Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-20T19:03:58.946312Z","submitted_at":"2025-07-25T12:39:03Z","title":"How Much Do Large Language Model Cheat on Evaluation? Benchmarking Overestimation under the One-Time-Pad-Based Framework"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":42,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 2 inbound Pith citation observations for arXiv:2507.19219."}