{"as_of":"2026-08-08T21:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:686373cd880e26972ee0db66f17a9d7f0c06ae976473311eafa493c9f5e92c2f","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T04:31:04.475518Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.02985/citation-record","integrity":"/paper/2608.02985/integrity","json":"/paper/2608.02985/citation-record.json","paper":"/paper/2608.02985"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.345601Z","title":"Look-ahead-bench: A standardized benchmark of look-ahead bias in point-in-time LLMs for finance.arXiv preprint arXiv:2601.13770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.345601Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:5ed23e269daa9755fed9226cfa0f9200413207aaea80239e320d45923c5c23d5","observation_id":"aa84cace-faa1-4012-bd2f-d475aca4e98c","resolution":{"observed_at":"2026-08-08T04:31:04.345601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.029410Z","title":null,"venue":null,"work_id":"8001f9e9-3436-4ad2-99b5-3492cb618a50","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.450644Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:ee78edc2365665faa9a470f528aeff67d2d9dbeb16bd0eadc7788f1404af066f","observation_id":"d6eff41d-17eb-4136-a1c6-7c3725515f50","resolution":{"observed_at":"2026-08-08T04:31:05.033066Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.975416Z","title":null,"venue":null,"work_id":"177cdfc5-cd8f-4581-a834-5fdb5d995164","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.469239Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:5af14ec6463d86e9ecafbe337b74dbb4bca07b30d04a140e27192a9972244a2f","observation_id":"2578518a-f8ae-4224-b19e-e50c39d6686a","resolution":{"observed_at":"2026-08-08T04:31:04.978959Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.087317Z","title":null,"venue":null,"work_id":"f3b6823f-18a7-4e88-b457-0f3cc5a3d9e4","year":1969},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.429286Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:3145f972bd7c602921f13907b3e2a62cedf2a01adf5f84c445224a117bef15c3","observation_id":"b21ca94e-9c29-49a0-85fc-b6201c56ca47","resolution":{"observed_at":"2026-08-08T04:31:05.090136Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09247","last_updated":"2024-10-11T20:46:56Z","snapshot_observed_at":"2026-07-06T19:32:06.379698Z","submitted_at":"2024-10-11T20:46:56Z","title":"Benchmark Inflation: Revealing LLM Performance Gaps Using Retro-Holdouts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.09247","snapshot_observed_at":"2026-08-08T04:31:04.364840Z","title":"Benchmark inflation: Revealing LLM performance gaps using retro-holdouts.arXiv preprint arXiv:2410.09247,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.364840Z"},"links":{"cited_paper":"/paper/2410.09247","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:413a7def6102da2fd94739a7bf6dd4373091fc017a0b22c1507daf2232294bfd","observation_id":"57e002fd-a558-4c7a-acf4-fe54ebfd2a4b","resolution":{"observed_at":"2026-08-08T04:31:04.364840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.069123Z","title":"Table 4 is the map: each theoretical claim of Sections 3 to 5 and the experiment whose headline result carries it","venue":null,"work_id":"269f0da9-619e-458f-8764-d67968b79aef","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.436187Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:d4f724f3a4aa1b7b78739d4c805dd62148b64531ef21aef038a119713df7669b","observation_id":"88949916-c22c-436b-aa7b-4bae0dce7452","resolution":{"observed_at":"2026-08-08T04:31:05.072640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.24564","last_updated":"2026-05-23T12:57:18Z","snapshot_observed_at":"2026-08-06T23:13:37.659266Z","submitted_at":"2026-05-23T12:57:18Z","title":"Summoning the Oracle to Slay It: Mitigating Look-Ahead Bias in Financial Backtesting with Large Language Models","version":1},"cited_work":{"arxiv_id":"2605.24564","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.24564","snapshot_observed_at":"2026-08-08T04:31:04.775797Z","title":"Summoning the Oracle to Slay It: Mitigating Look-Ahead Bias in Financial Backtesting with Large Language Models","venue":"cs.AI","work_id":"957796e2-fade-4440-bd7f-33d3827077f1","year":2026},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.376212Z"},"links":{"cited_paper":"/paper/2605.24564","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:d0c0d28238fc33a6035376f64a6c21316ac55be43c6cb4656c5b3911808a0905","observation_id":"62b7d53d-17d2-42c7-90eb-eefe20d0813a","resolution":{"observed_at":"2026-08-08T04:31:04.780051Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19533","last_updated":"2025-05-26T05:39:57Z","snapshot_observed_at":"2026-08-08T00:00:45.680749Z","submitted_at":"2025-05-26T05:39:57Z","title":"ExAnte: A Benchmark for Ex-Ante Inference in Large Language Models","version":1},"cited_work":{"arxiv_id":"2505.19533","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.19533","snapshot_observed_at":"2026-08-08T04:31:04.760758Z","title":"ExAnte: A Benchmark for Ex-Ante Inference in Large Language Models","venue":"cs.LG","work_id":"52567b85-577a-479d-be2f-fb234b1b2eac","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.379362Z"},"links":{"cited_paper":"/paper/2505.19533","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:8b3ad52d26f247948d29a38109eeb38465e0621704dcb17f2ee21c43c557c4fe","observation_id":"34e6a539-1707-4c8c-83d1-cb68f3072666","resolution":{"observed_at":"2026-08-08T04:31:04.765917Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17623","last_updated":"2023-11-24T01:45:16Z","snapshot_observed_at":"2026-07-06T16:39:08.850847Z","submitted_at":"2023-10-26T17:43:13Z","title":"Proving Test Set Contamination in Black Box Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17623","snapshot_observed_at":"2026-08-08T04:31:04.382539Z","title":"Daniel Paleka, Shashwat Goel, Jonas Geiping, and Florian Tramèr","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.382539Z"},"links":{"cited_paper":"/paper/2310.17623","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:ec395cd0de84f229258edfc90fe59d9f84fdf2663862b1eb596b362ca9d66af7","observation_id":"f10255f9-bef7-4875-8408-e3c823662b4e","resolution":{"observed_at":"2026-08-08T04:31:04.382539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.00723","last_updated":"2025-05-31T21:49:17Z","snapshot_observed_at":"2026-08-07T11:57:07.802306Z","submitted_at":"2025-05-31T21:49:17Z","title":"Pitfalls in Evaluating Language Model Forecasters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.00723","snapshot_observed_at":"2026-08-08T04:31:04.385535Z","title":"14 Andrew J Patton and Allan Timmermann","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.385535Z"},"links":{"cited_paper":"/paper/2506.00723","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:750c0a3be1a215d7338e51991022340fee1f901661a87b2f0c1b631670608b72","observation_id":"7f3bb58b-25dc-4411-b7fa-83f44904762a","resolution":{"observed_at":"2026-08-08T04:31:04.385535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.00699","last_updated":"2025-07-09T19:56:02Z","snapshot_observed_at":"2026-07-06T17:53:40.016965Z","submitted_at":"2024-03-31T14:32:02Z","title":"A Comprehensive Survey of Contamination Detection Methods in Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.00699","snapshot_observed_at":"2026-08-08T04:31:04.388470Z","title":"Martin Riddell, Ansong Ni, and Arman Cohan","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.388470Z"},"links":{"cited_paper":"/paper/2404.00699","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:841a15896b1c0bc41ad4f3e4b7b920dcb6b009fa42076e4f0ec6d714e674a551","observation_id":"c1155d93-b9f5-4885-bd2a-7d502bc2c26c","resolution":{"observed_at":"2026-08-08T04:31:04.388470Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04811","last_updated":"2024-03-06T21:45:35Z","snapshot_observed_at":"2026-07-06T17:41:14.805521Z","submitted_at":"2024-03-06T21:45:35Z","title":"Quantifying Contamination in Evaluating Code Generation Capabilities of Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04811","snapshot_observed_at":"2026-08-08T04:31:04.391835Z","title":"Manley Roberts, Himanshu Thakur, Christine Herlihy, Colin White, and Samuel Dooley","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.391835Z"},"links":{"cited_paper":"/paper/2403.04811","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:4cace24602f6495ce9e3a5cdce5d9ffe839d5c63ae834a5340dbbc5bb200a7bb","observation_id":"765eb102-f051-4d11-bad6-a9e8a45ace8c","resolution":{"observed_at":"2026-08-08T04:31:04.391835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.095995Z","title":"NLP evaluation in trouble: On the need to measure LLM data contamination for each benchmark","venue":null,"work_id":"7c97fed7-9e6c-46fd-b144-f06d9f74e72d","year":2023},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.395288Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:f43f36114b52a67b5b626e7f977db5281d5dbc8c5220ab5684abf2da56b80a20","observation_id":"7d15f498-479f-4ef2-a62c-23e136f7c027","resolution":{"observed_at":"2026-08-08T04:31:05.099410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.398711Z","title":"Quantifying the effect of test set contamination on generative evaluations.arXiv preprint arXiv:2601.04301,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.398711Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:d26dcc9549a2e3ae1db0fc2facde9353081c893ebe570bb71a4ea3f16c2f0413","observation_id":"853a2f93-9126-4740-8770-3eedf6e9fa71","resolution":{"observed_at":"2026-08-08T04:31:04.398711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.405329Z","title":"Colin White, Samuel Dooley, Manley Roberts, Arka Pal, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.405329Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:5603c1e12592985dadef8007c4f9dd8afeabaa78ceba84080da70edcd11ef27b","observation_id":"bc60c097-8936-407a-b34d-3e600514dff0","resolution":{"observed_at":"2026-08-08T04:31:04.405329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-08T04:31:04.408456Z","title":"Jeffrey M Wooldridge","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.408456Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:c968b396437a649ff79df7d6c9f109e828fae5e67352334f0bfb70414068d177","observation_id":"d4342b7b-c2a5-44d1-bc7c-7682006aeb82","resolution":{"observed_at":"2026-08-08T04:31:04.408456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04244","last_updated":"2024-06-06T16:41:39Z","snapshot_observed_at":"2026-07-30T15:43:06.151242Z","submitted_at":"2024-06-06T16:41:39Z","title":"Benchmark Data Contamination of Large Language Models: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04244","snapshot_observed_at":"2026-08-08T04:31:04.411806Z","title":"Benchmark data contamination of large language models: A survey.arXiv preprint arXiv:2406.04244,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.411806Z"},"links":{"cited_paper":"/paper/2406.04244","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:79a616f306e42707dae8fd67b9226ef1732a91c277f03a6662ff3da2955ad2b7","observation_id":"772e1434-6b8e-4cfa-937d-653c40d82d9d","resolution":{"observed_at":"2026-08-08T04:31:04.411806Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.11838","last_updated":"2026-07-23T15:24:41Z","snapshot_observed_at":"2026-08-07T10:07:46.710244Z","submitted_at":"2026-03-12T12:04:43Z","title":"DatedGPT: Preventing Lookahead Bias in Large Language Models with Time-Aware Pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.11838","snapshot_observed_at":"2026-08-08T04:31:04.415388Z","title":"DatedGPT: Preventing lookahead bias in large language models with time-aware pretraining.arXiv preprint arXiv:2603.11838,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.415388Z"},"links":{"cited_paper":"/paper/2603.11838","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:d28729ebfdd8ce44d9cfb72d6351b995d4172cfee2bb77e7b5c00e09efdd01de","observation_id":"b6d5b39f-7720-483f-a62c-bd18e18337aa","resolution":{"observed_at":"2026-08-08T04:31:04.415388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04850","last_updated":"2023-11-11T05:11:18Z","snapshot_observed_at":"2026-08-07T11:21:12.881808Z","submitted_at":"2023-11-08T17:35:20Z","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04850","snapshot_observed_at":"2026-08-08T04:31:04.418737Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.418737Z"},"links":{"cited_paper":"/paper/2311.04850","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:e20c460f9083d065f5a9c962a71c81cd0a4fc2ddbdf7ab747f3cf150918ae6a1","observation_id":"372bad00-77da-4547-981f-6351c5a94418","resolution":{"observed_at":"2026-08-08T04:31:04.418737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.27055","last_updated":"2026-05-12T17:08:27Z","snapshot_observed_at":"2026-07-06T22:34:33.419178Z","submitted_at":"2025-10-30T23:50:05Z","title":"Detecting Data Contamination in LLMs via In-Context Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.27055","snapshot_observed_at":"2026-08-08T04:31:04.422329Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.422329Z"},"links":{"cited_paper":"/paper/2510.27055","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:9177e91027662351d58a0897ddc44fcbb18fa9a45caa021584a5da17b8031a5c","observation_id":"41e2cb85-5083-426e-b99e-3fe4ba1df6b5","resolution":{"observed_at":"2026-08-08T04:31:04.422329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.00332","last_updated":"2024-11-22T22:27:49Z","snapshot_observed_at":"2026-07-06T18:08:04.815730Z","submitted_at":"2024-05-01T05:52:05Z","title":"A Careful Examination of Large Language Model Performance on Grade School Arithmetic","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.00332","snapshot_observed_at":"2026-08-08T04:31:04.425762Z","title":"Zeyu Zhang, Ryan Chen, and Bradly C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.425762Z"},"links":{"cited_paper":"/paper/2405.00332","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:35195b9bcc36ca0b3cd409a020831f4bcd2a8c45e8663da8cf9a4ddd1b0d546b","observation_id":"cf2f86c7-c6f3-492d-bfa4-9deb7ac76e34","resolution":{"observed_at":"2026-08-08T04:31:04.425762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.060175Z","title":null,"venue":null,"work_id":"3776cee5-1332-4109-ae4c-91de06552921","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.439874Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:ee8ae211f39f0d26f06947e7240a7a2bb933b542f257a1f2790b3b77614f6cd2","observation_id":"8b6905b3-5d1a-4114-b422-92d06516544b","resolution":{"observed_at":"2026-08-08T04:31:05.063406Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.049937Z","title":null,"venue":null,"work_id":"75e733aa-219c-4f59-be1e-727f73fa4f63","year":2026},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.443472Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:6eb05cb483838bb487ade040747be81880b32b95ab9f7b96d612553ce1d6af71","observation_id":"db180450-b928-4133-9961-150004c49a3a","resolution":{"observed_at":"2026-08-08T04:31:05.053534Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.039551Z","title":null,"venue":null,"work_id":"52b7ea57-c5a5-41d5-b8a1-76ed4a482bb2","year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.447050Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:63bb8ae2add263a6bef9a684a47a033e5145c952c5551acd9dc2ca4a269b0ec3","observation_id":"59330655-316d-4c44-b6a5-75ccfaa33fd2","resolution":{"observed_at":"2026-08-08T04:31:05.043282Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.019057Z","title":"Controls are MiniMax-M3 and Claude-Opus-4.7, whose January 2026 cutoffs leave no leakage discontinuity inside the tested window","venue":null,"work_id":"7ab03062-4ff4-480d-afc4-c631baf4c119","year":2026},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.453942Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:0d73220cd9f5aaf5c38d593f7371e2be5ce8434c01e9114a01d875f06381dc16","observation_id":"38c1209c-35c2-409b-8cae-1b0c3fa5eb29","resolution":{"observed_at":"2026-08-08T04:31:05.022778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.997066Z","title":"Eachproblemcarriesitscontest release date, so a model can only have trained on a problem’s solution if the contest occurred before the model’s training cutoff","venue":null,"work_id":"355d0c99-67e3-4d2a-86df-328e87c8fb52","year":2023},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.461264Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:bf3edbb7aa9f500490cf5113472ead70c96a8be70f2681e4c3efb83fab6ec8e0","observation_id":"e5bd1bdb-b5eb-4a63-841f-2cd158db3496","resolution":{"observed_at":"2026-08-08T04:31:05.000807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.964775Z","title":"paraphrases","venue":null,"work_id":"f8696eb0-bfae-40cf-bb4b-73eda8923c3b","year":2000},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.472308Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:797cf4986af778a1df8fa9902eef98d54921af8a42a608584f1c016deecb2b9f","observation_id":"9e29a8fe-3126-4a30-9921-726385db38a5","resolution":{"observed_at":"2026-08-08T04:31:04.968317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.952137Z","title":"Treatment–control PRC excess by domain and pooled, under Platt and isotonic calibration","venue":null,"work_id":"331c1aa0-127b-4525-b571-8545faaaccea","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.475518Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:a9b76c8eeb91d95c00a2d4ca2e9ab3b3558a1bbbc57bd725fffa52f5f65e6aa9","observation_id":"1686ff19-2009-461f-a0d6-df65bd193856","resolution":{"observed_at":"2026-08-08T04:31:04.955926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.007847Z","title":"Because the leakage estimand is ajumprather than a level, protocol level effects cancel unless they vary sharply in time","venue":null,"work_id":"a5c2ec15-0257-4135-9375-a97cbf36c566","year":2025},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":910,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.457610Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:01ff9efd4549b7490af02c570a947948c1cf7f59dbf389c6b2bdebec5222ab6d","observation_id":"e8e041c5-3ade-46a3-b70c-5f8d4a85de48","resolution":{"observed_at":"2026-08-08T04:31:05.011782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21206","last_updated":"2025-07-06T01:02:58Z","snapshot_observed_at":"2026-08-07T17:39:15.439590Z","submitted_at":"2025-02-28T16:25:50Z","title":"Chronologically Consistent Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.21206","snapshot_observed_at":"2026-08-08T04:31:04.368666Z","title":"Chronologically consistent large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":1978,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.368666Z"},"links":{"cited_paper":"/paper/2502.21206","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:2a08d9fc558e2e7984bc24dca36c2f68f74488ea2af7de6a95a3dffe279cf516","observation_id":"65fe6454-1ca9-49f2-83ea-6f93f1670b22","resolution":{"observed_at":"2026-08-08T04:31:04.368666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.361204Z","title":"Detecting lookahead bias in LLM forecasts.arXiv preprint arXiv:2512.23847,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":1987,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.361204Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:c9cddcfe30bc60fb6ed06ff5f9c9c61984b8433478ec222160e2d6650e437eec","observation_id":"c79b8ee5-2332-4943-a681-58e9ec2b1c02","resolution":{"observed_at":"2026-08-08T04:31:04.361204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.078297Z","title":"The sharp-bounds framing of Theorem 2 follows partial identification (Manski, 2003; Imbens & Manski, 2004)","venue":null,"work_id":"f2b8c5ac-a3f7-4d7f-b847-554a139deca4","year":2015},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2010,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.432684Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:9b9cc0f9a89e3c974b99fdc187cc98e766375b75b280bed6c7bcfaa6398aaf16","observation_id":"5faa8654-e29e-4563-88e5-c5054eb22f46","resolution":{"observed_at":"2026-08-08T04:31:05.081622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03923","last_updated":"2024-11-06T13:54:08Z","snapshot_observed_at":"2026-08-07T09:59:12.000905Z","submitted_at":"2024-11-06T13:54:08Z","title":"Evaluation data contamination in LLMs: how do we measure it and (when) does it matter?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03923","snapshot_observed_at":"2026-08-08T04:31:04.401902Z","title":"Singh, Muhammed Yusuf Kocyigit, Andrew Poulton, David Esiobu, Maria Lomeli, Gergely Szilvasy, and Dieuwke Hupkes","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.401902Z"},"links":{"cited_paper":"/paper/2411.03923","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:559968bc17be1b0d6f67b77f18ad59fe3263b1795d502a6a312864fdcd23cb25","observation_id":"c0328c06-8a03-4415-88fd-e754b03f040c","resolution":{"observed_at":"2026-08-08T04:31:04.401902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.105356Z","title":"Time machine GPT","venue":null,"work_id":"388387dd-b677-42a9-a1b4-d151f487cd63","year":2024},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.353655Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:0679f11f8a68735342006e6594e3e5435ea6708c5bbb826bec4e7173e61ddba9","observation_id":"a0004f2f-ec5b-48d7-ae3c-becedf7c2733","resolution":{"observed_at":"2026-08-08T04:31:05.108653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:04.985895Z","title":"Composition controls have cutoffsafterthe latest problem (Gemini-2.5-Pro, DeepSeek-R1, and the published 2025-cutoff pool)","venue":null,"work_id":"8cc3daa2-324b-438f-8b62-2da20c9eb099","year":2026},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.465422Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:c02b9b1b0367277b97ec4a7c7931ee01ed7f94fffd61515ff96764258091c32a","observation_id":"5028d306-cf6b-4a91-acca-f6e123629764","resolution":{"observed_at":"2026-08-08T04:31:04.990275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07841","last_updated":"2024-09-16T13:18:23Z","snapshot_observed_at":"2026-08-07T03:58:21.441544Z","submitted_at":"2024-02-12T17:52:05Z","title":"Do Membership Inference Attacks Work on Large Language Models?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07841","snapshot_observed_at":"2026-08-08T04:31:04.357384Z","title":"Graham Elliott and Allan Timmermann.Economic Forecasting","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.357384Z"},"links":{"cited_paper":"/paper/2402.07841","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:4d04e800e41f8dd3925e0b5ea067523e29324e542b280d0363c03b3e18aca6bc","observation_id":"3b6506ba-f4ae-43db-9bd6-7e0531d18027","resolution":{"observed_at":"2026-08-08T04:31:04.357384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-08-08T04:31:04.372400Z","title":"Minhao Jiang, Ken Ziyu Liu, Ming Zhong, Rylan Schaeffer, Siru Ouyang, Jiawei Han, and Sanmi Koyejo","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.372400Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:beb18df6d0f074fcef9df72ea6a4399f19688b5f63a8ed23e92696e6be4da8f3","observation_id":"8ba30b76-3c39-4208-82c9-d9ee8bfc293e","resolution":{"observed_at":"2026-08-08T04:31:04.372400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T04:31:05.115126Z","title":"Language models are few-shot learners","venue":null,"work_id":"81c0a065-7e72-487a-a181-60248a4d06b3","year":1901},"citing_paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-08T04:31:04.349752Z"},"links":{"citing_paper":"/paper/2608.02985"},"observation_digest":"sha256:684b6ddc5c21caf9917d68645d5d4a022e17e37d39d0f3170e6af9993a4d41bc","observation_id":"f3c8b989-f5b5-40e6-8984-7503679df9e1","resolution":{"observed_at":"2026-08-08T04:31:05.118711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.02985","last_updated":"2026-08-04T00:45:54Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-08T04:19:18.723440Z","submitted_at":"2026-08-04T00:45:54Z","title":"Temporal Leakage in LLM Backtesting: Measurement, Validation, and Adjusted Scores"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":25,"verified_exact":1,"verified_fuzzy":11},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 0 inbound Pith citation observations for arXiv:2608.02985."}