{"as_of":"2026-08-08T21:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b2b56740b4609ed815e5df21f61387795a35d8dae3b6600b2b5980f429bcbf3c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":20,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T14:51:15.453440Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":8,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-08T14:51:15.453440Z","title":"A., Garc ´ıa-Ferrero, I., Etxaniz, J., de Lacalle, O","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06655","last_updated":"2025-05-12T14:34:05Z","snapshot_observed_at":"2026-08-08T14:44:11.158498Z","submitted_at":"2025-02-10T16:45:18Z","title":"Unbiased Evaluation of Large Language Models from a Causal Perspective","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T14:51:15.453440Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2502.06655"},"observation_digest":"sha256:ffc1d529fb3cf16313e315fdfb93c49ac787c6937c9602df1038fc78623dc60a","observation_id":"d38930be-bc42-42ce-8362-ea6b0e0573df","resolution":{"observed_at":"2026-08-08T14:51:15.453440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-07T11:04:07.003448Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.03637","last_updated":"2025-07-07T09:53:22Z","snapshot_observed_at":"2026-08-08T03:23:19.289933Z","submitted_at":"2025-06-04T07:30:16Z","title":"RewardAnything: Generalizable Principle-Following Reward Models","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T11:04:07.003448Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2506.03637"},"observation_digest":"sha256:12026719930e786a50b63619732de9cc64ed8748e38e4f9ba0a30b1502c5cb50","observation_id":"9171a74f-991e-4ff4-b18d-8ed9ca0db66c","resolution":{"observed_at":"2026-08-07T11:04:07.003448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-07T10:54:26.345390Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04142","last_updated":"2025-06-04T16:33:44Z","snapshot_observed_at":"2026-08-07T21:43:19.257599Z","submitted_at":"2025-06-04T16:33:44Z","title":"Establishing Trustworthy LLM Evaluation via Shortcut Neuron Analysis","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T10:54:26.345390Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2506.04142"},"observation_digest":"sha256:ed7dfee2c33d4a81836ef268dbe02faa3dbc224c849cfd02f257b40a3f3d2faa","observation_id":"6ff839f5-66d4-4b38-947c-22946783829d","resolution":{"observed_at":"2026-08-07T10:54:26.345390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2507.22359","last_updated":"2026-04-14T11:47:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-30T03:50:46Z","title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","version":4},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-19T03:17:06.457421Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2507.22359"},"observation_digest":"sha256:7304c96cb194e0e3bc421cdd64c0a4b5de7fc5ba3f77364c75082d686ef80ba5","observation_id":"67128382-2d9a-44be-83dd-fd934e1ae9bd","resolution":{"observed_at":"2026-05-19T03:22:01.356975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T19:28:54.967333Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.12464","last_updated":"2025-08-17T18:27:54Z","snapshot_observed_at":"2026-08-06T19:31:58.152011Z","submitted_at":"2025-08-17T18:27:54Z","title":"On the Fitness Landscape in the $NK$ Model","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T19:28:54.967333Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2508.12464"},"observation_digest":"sha256:b0434bf96687ad91fff35361bde6a5513d4a1a1150280626a30d682dc82b0c33","observation_id":"0f330cad-9e76-467d-95ef-727506a63474","resolution":{"observed_at":"2026-08-05T19:28:54.967333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2509.23108","last_updated":"2026-05-19T05:14:38Z","snapshot_observed_at":"2026-07-06T22:30:55.313733Z","submitted_at":"2025-09-27T04:36:12Z","title":"Artificial Phantasia: Emergent Mental Imagery in Large Language Models","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-21T21:41:39.111769Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2509.23108"},"observation_digest":"sha256:2c75b1b06f06cfbea960321ef07b43859dd0fdb9bbd08304c18829f671e5753e","observation_id":"56c0d6cd-85ae-444a-96c4-7198018eabf1","resolution":{"observed_at":"2026-05-21T21:44:22.869054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-04T11:19:49.706763Z","title":"NLP evaluation in trouble: On the need to measure LLM data contamination for each benchmark, December 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.05709","last_updated":"2026-06-04T12:15:57Z","snapshot_observed_at":"2026-08-07T08:05:26.965055Z","submitted_at":"2025-10-07T09:22:22Z","title":"Correcting Prompt Dependence in LLM Benchmarks: A Bayesian Hierarchical Model with Embedding-Space Clustering","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-04T11:19:49.706763Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2510.05709"},"observation_digest":"sha256:8a1dac7d18893d344997fb6ebb8d1ba560f8ae8e68da45cf2c2348ceed9bd9e9","observation_id":"e5b6228e-fa83-4191-b596-db823287d718","resolution":{"observed_at":"2026-08-04T11:19:49.706763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2511.02627","last_updated":"2026-06-11T15:36:40Z","snapshot_observed_at":"2026-08-04T00:11:48.936857Z","submitted_at":"2025-11-04T14:57:11Z","title":"DecompSR: A dataset for decomposed analyses of compositional multihop spatial reasoning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T01:18:44.523602Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2511.02627"},"observation_digest":"sha256:f293f33a9bb9a1cfecd174dafedc08e8cf054ccc81657f0e9200d7e5b28a1456","observation_id":"293cbe29-27d6-4e41-91e6-87ae083e33c7","resolution":{"observed_at":"2026-05-18T01:20:34.337660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-04T00:11:51.863397Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.02627","last_updated":"2026-06-11T15:36:40Z","snapshot_observed_at":"2026-08-04T00:11:48.936857Z","submitted_at":"2025-11-04T14:57:11Z","title":"DecompSR: A dataset for decomposed analyses of compositional multihop spatial reasoning","version":4},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-04T00:11:51.863397Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2511.02627"},"observation_digest":"sha256:0b7999f1b52d88ada35df7716dfc64e8abf6af9fcc2bc41009357fea2fb4bc3d","observation_id":"5fe6ffde-e7d3-4b18-9162-da15dd263a19","resolution":{"observed_at":"2026-08-04T00:11:51.863397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-03T06:46:26.337035Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark.arXiv preprint arXiv:2310.18018, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.22025","last_updated":"2026-06-09T23:57:32Z","snapshot_observed_at":"2026-08-07T21:40:38.906811Z","submitted_at":"2026-01-29T17:32:34Z","title":"When Generic Prompt Improvements Hurt: Evaluation-Driven Iteration for LLM Applications","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T06:46:26.337035Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2601.22025"},"observation_digest":"sha256:bae1c5780c9c1cb340fb4a5705be25b1b5ee69569fff0a578af89167a54442aa","observation_id":"d6048266-81af-4ed5-9c79-d42da57ac594","resolution":{"observed_at":"2026-08-03T06:46:26.337035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2604.18955","last_updated":"2026-04-21T01:05:52Z","snapshot_observed_at":"2026-07-06T23:05:44.499005Z","submitted_at":"2026-04-21T01:05:52Z","title":"Assessing Capabilities of Large Language Models in Social Media Analytics: A Multi-task Quest","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-10T03:20:12.777878Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2604.18955"},"observation_digest":"sha256:e23a75862b683276ec6b3ac03b9523bcc2d7e74a4c1f9c939cf8d19e7bd788d3","observation_id":"14e051e7-4ac8-4f34-8c88-9452415addd0","resolution":{"observed_at":"2026-05-11T12:41:01.820540Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2604.20273","last_updated":"2026-04-22T07:20:03Z","snapshot_observed_at":"2026-07-06T23:06:44.624357Z","submitted_at":"2026-04-22T07:20:03Z","title":"ActuBench: A Multi-Agent LLM Pipeline for Generation and Evaluation of Actuarial Reasoning Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T00:35:24.397273Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2604.20273"},"observation_digest":"sha256:915fe68fb5fc70d17a964224b25ca4bd6fae80ad5b9925507e30c3d261a35fe3","observation_id":"0c62ac00-20a6-47ee-9c7c-338a0ba8f1a1","resolution":{"observed_at":"2026-05-11T13:46:05.187194Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2605.24213","last_updated":"2026-05-22T20:54:30Z","snapshot_observed_at":"2026-08-02T06:58:01.300783Z","submitted_at":"2026-05-22T20:54:30Z","title":"Towards Evaluation Engineering: An Empirical Study of ML Evaluation Harnesses in the Wild","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-30T14:41:07.354007Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2605.24213"},"observation_digest":"sha256:bc2508e11791d4b4a65f05a174929d8e9cbf2ea35c165526573920bda7379e88","observation_id":"69286f6b-21dd-4371-9c4e-7f613ba6fab7","resolution":{"observed_at":"2026-06-30T14:44:45.135213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2605.26133","last_updated":"2026-05-21T10:32:33Z","snapshot_observed_at":"2026-08-07T10:44:40.983544Z","submitted_at":"2026-05-21T10:32:33Z","title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-30T17:20:16.735285Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2605.26133"},"observation_digest":"sha256:2c103a62de27afbe40d2c2acfad3691653494369f280fc5c08db4324b67c1447","observation_id":"66eb410d-85f7-4e23-b259-15edd22a1667","resolution":{"observed_at":"2026-06-30T17:24:56.613744Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2605.26781","last_updated":"2026-05-26T09:50:35Z","snapshot_observed_at":"2026-07-06T23:36:34.652096Z","submitted_at":"2026-05-26T09:50:35Z","title":"LiveK12Bench: Have Large Multimodal Models Truly Conquered High School-level Examinations?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T17:33:03.397468Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2605.26781"},"observation_digest":"sha256:85f0fd261d96ac6493ec774ff397f10243e044b8a2e8b083e2e1410b8b4f670c","observation_id":"8bbe3091-2d39-4119-8607-e5f378ccfd67","resolution":{"observed_at":"2026-06-29T17:33:45.019663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2606.01189","last_updated":"2026-05-31T12:11:47Z","snapshot_observed_at":"2026-08-06T00:50:49.461826Z","submitted_at":"2026-05-31T12:11:47Z","title":"The Case for Model Science: Verify, Explore, Steer, Refine","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-28T17:24:32.311565Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2606.01189"},"observation_digest":"sha256:50b5dd6dda029334484a78338d975130a4d22ecf3794ca0481b71077767b1b33","observation_id":"8ba959b7-1a87-4e52-8441-eaf435884bdf","resolution":{"observed_at":"2026-07-01T21:16:13.563798Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2606.17454","last_updated":"2026-06-17T04:51:06Z","snapshot_observed_at":"2026-08-02T05:33:09.695431Z","submitted_at":"2026-06-16T03:17:03Z","title":"Dissecting model behavior through agent trajectories","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T01:27:39.812496Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2606.17454"},"observation_digest":"sha256:379a4ec2c80ad0291f4592b16a7505d666bd4e7c37d845d079810b82c3b21576","observation_id":"66ad5f14-c0ad-47f2-80ba-698b5e72f080","resolution":{"observed_at":"2026-07-03T20:18:56.917233Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2607.00276","last_updated":"2026-06-30T23:52:15Z","snapshot_observed_at":"2026-08-08T10:47:33.889635Z","submitted_at":"2026-06-30T23:52:15Z","title":"Testing Frontier Large Language Models' Physics Literacy in Parallel Physical Worlds","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-07-02T19:18:43.558804Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2607.00276"},"observation_digest":"sha256:09336840740cf572eba8f017eb6b5521c597f17105411aac3ab10a4dbe4009ee","observation_id":"17835370-b4a7-4b61-bee8-ff4db2e32631","resolution":{"observed_at":"2026-07-02T19:27:18.679688Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":"2310.18018","doi":"10.48550/arxiv.2310.18018","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp evaluation in trouble: On the need to measure llm data contamination for each benchmark","venue":"arXiv (Cornell University)","work_id":"feaf47be-d93d-4603-a19e-9dfb3378c90c","year":2023},"citing_paper":{"arxiv_id":"2607.01829","last_updated":"2026-07-02T07:49:55Z","snapshot_observed_at":"2026-07-07T00:07:19.026006Z","submitted_at":"2026-07-02T07:49:55Z","title":"Pre-Flight: A Benchmark for Evaluating Large Language Models on Aviation Operational Knowledge","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-03T13:36:31.189451Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2607.01829"},"observation_digest":"sha256:80ef7136e5e34ff313ef426d211cf5b6f2598b226b329d058b6280e5a1f74923","observation_id":"4482a391-6e89-4f76-9714-9d72df27b199","resolution":{"observed_at":"2026-07-03T13:38:18.445574Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18018","snapshot_observed_at":"2026-08-02T13:26:26.237664Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark, October 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19355","last_updated":"2026-05-22T12:13:16Z","snapshot_observed_at":"2026-08-07T21:44:05.596114Z","submitted_at":"2026-05-22T12:13:16Z","title":"Information Discernment in Large Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T13:26:26.237664Z"},"links":{"cited_paper":"/paper/2310.18018","citing_paper":"/paper/2607.19355"},"observation_digest":"sha256:3e08ca9f04c987faea3bbf91b737d3373eea38966b42e276cad5b95462278eac","observation_id":"f07d090d-375a-4152-ad26-b306a029ab47","resolution":{"observed_at":"2026-08-02T13:26:26.237664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.18018/citation-record","integrity":"/paper/2310.18018/integrity","json":"/paper/2310.18018/citation-record.json","paper":"/paper/2310.18018"},"outbound":[],"paper":{"arxiv_id":"2310.18018","last_updated":"2023-10-27T09:48:29Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T21:43:48.165854Z","submitted_at":"2023-10-27T09:48:29Z","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 20 inbound Pith citation observations for arXiv:2310.18018."}