{"as_of":"2026-08-05T20:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:23aed85e6e1a58138ceaee6a2ab7f1de7d5cea7a2aea355eefaa8cf35b722705","coverage":[{"denominator":5,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-08T03:30:55.419225Z","state":"measured"},{"denominator":10,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":10,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T14:53:54.893111Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-14T20:32:56.827813Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"cited_work":{"arxiv_id":"2604.24955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.24955","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","venue":"cs.CL","work_id":"9d211200-800c-4969-9643-3b694c6111c8","year":2026},"citing_paper":{"arxiv_id":"2604.27977","last_updated":"2026-05-01T17:11:52Z","snapshot_observed_at":"2026-07-31T18:07:13.372491Z","submitted_at":"2026-04-30T15:06:56Z","title":"D3-Gym: Constructing Real-World Verifiable Environments for Data-Driven Discovery","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2604.24955","citing_paper":"/paper/2604.27977"},"observation_digest":"sha256:06424e4967baf36617325e4fce23db9c0ec3fce7ef3053e8c3f42edfbf30c591","observation_id":"0ab24ec6-9037-4b5e-9bf8-233154454a43","resolution":{"observed_at":"2026-05-09T04:50:11.921338Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"cited_work":{"arxiv_id":"2604.24955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.24955","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","venue":"cs.CL","work_id":"9d211200-800c-4969-9643-3b694c6111c8","year":2026},"citing_paper":{"arxiv_id":"2605.04624","last_updated":"2026-07-24T02:31:26Z","snapshot_observed_at":"2026-08-02T14:53:50.130923Z","submitted_at":"2026-05-06T08:12:09Z","title":"AuditRepairBench: A Paired-Execution Trace Corpus for Evaluator-Channel Ranking Instability in Agent Repair","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-08T17:51:54.468688Z"},"links":{"cited_paper":"/paper/2604.24955","citing_paper":"/paper/2605.04624"},"observation_digest":"sha256:024553bc6194ea9ae11fb250ae4347e0b28d4869936a11223cc1922ecb7a1f7a","observation_id":"5952ca91-70db-4f34-acdd-d10eddcb4323","resolution":{"observed_at":"2026-05-11T17:11:07.694606Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.24955","snapshot_observed_at":"2026-08-02T14:53:54.893111Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.04624","last_updated":"2026-07-24T02:31:26Z","snapshot_observed_at":"2026-08-02T14:53:50.130923Z","submitted_at":"2026-05-06T08:12:09Z","title":"AuditRepairBench: A Paired-Execution Trace Corpus for Evaluator-Channel Ranking Instability in Agent Repair","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T14:53:54.893111Z"},"links":{"cited_paper":"/paper/2604.24955","citing_paper":"/paper/2605.04624"},"observation_digest":"sha256:d6eb08fd76bbdd5ffec93effba2556b53ed723b5bd6470b1d413767c05b05150","observation_id":"1274377b-70ea-4238-85da-811204810912","resolution":{"observed_at":"2026-08-02T14:53:54.893111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"cited_work":{"arxiv_id":"2604.24955","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.24955","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","venue":"cs.CL","work_id":"9d211200-800c-4969-9643-3b694c6111c8","year":2026},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-02T02:31:32.394072Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2604.24955","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:f93c97827df10821b70111e540b13eedc8c105ce0a58966c0f32a47b2559d993","observation_id":"f8358352-54c7-4d15-a484-b46f2d7741a9","resolution":{"observed_at":"2026-05-14T20:32:56.830494Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.24955","snapshot_observed_at":"2026-08-01T06:24:59.579784Z","title":"Bench- Guard: Who guards the benchmarks? automated auditing of LLM agent benchmarks.arXiv preprint arXiv:2604.24955,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27518","last_updated":"2026-07-29T23:17:08Z","snapshot_observed_at":"2026-08-02T23:33:51.735940Z","submitted_at":"2026-07-29T23:17:08Z","title":"Automated Transcript Analysis for Detecting Flaws in Agentic Benchmarks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T06:24:59.579784Z"},"links":{"cited_paper":"/paper/2604.24955","citing_paper":"/paper/2607.27518"},"observation_digest":"sha256:bb79789ba5204d59ee24bb6b1a63c2e45aa78454ad5eaed3cf8fddd93679a9ae","observation_id":"232bb22d-d1e1-423c-958b-43a7223beb6b","resolution":{"observed_at":"2026-08-01T06:24:59.579784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.24955/citation-record","integrity":"/paper/2604.24955/integrity","json":"/paper/2604.24955/citation-record.json","paper":"/paper/2604.24955"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1086/261651","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ziru Chen, Shijie Chen, Yuting Ning, Qianheng Zhang, Boshi Wang, Botao Yu, Yifei Li, Zeyi Liao, Chen Wei, Zitong Lu, Vishal Dey, Mingyi Xue, Frazier N","venue":"Journal of Political Economy","work_id":"6a31fcc5-d7ee-4d75-95a9-3449fcdd54c9","year":null},"citing_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T03:30:55.419225Z"},"links":{"citing_paper":"/paper/2604.24955"},"observation_digest":"sha256:587c1e5a118dc376149a3a0dd8cf650a49645ed3c003f82b120cfa2f29ab50b7","observation_id":"bb5f9a46-0b98-4801-b6f3-9316f41b18e0","resolution":{"observed_at":"2026-05-09T00:24:28.380005Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.naacl-long.262","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proceedings of the 2025","venue":null,"work_id":"fe122ae3-cc68-452c-9514-8c315e0359c3","year":2025},"citing_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T03:30:55.419225Z"},"links":{"citing_paper":"/paper/2604.24955"},"observation_digest":"sha256:b0b6477dd95f546021589723916992c9179f0dbb8e4dc2b41c182f5f885ad480","observation_id":"8a715fd2-2bac-4910-894c-15bc19f02fca","resolution":{"observed_at":"2026-05-09T00:24:28.387336Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-23T14:53:17.606742+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T14:53:17.606742+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/0142-694x(91)90003-f","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:57:06.630808Z","title":"Jansson and Steven M","venue":"Design Studies","work_id":"f95d9bbf-626f-4d17-af02-1b505ce79392","year":1991},"citing_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-08T03:30:55.419225Z"},"links":{"citing_paper":"/paper/2604.24955"},"observation_digest":"sha256:4096b917b4c891893103454f86e20424626e4b250aae5c5468ee3e0d97be1d36","observation_id":"8ac8b491-a2b7-4f4f-aff5-06f596c06d5f","resolution":{"observed_at":"2026-05-09T00:24:28.383454Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.248","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","venue":null,"work_id":"cf61d957-e8b4-4b70-b1b5-ff4f7d3d8f0f","year":2024},"citing_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-08T03:30:55.419225Z"},"links":{"citing_paper":"/paper/2604.24955"},"observation_digest":"sha256:33cafc2780af82ac4abe8bfbefde8385eec6a455d4a3f049de870aa719e2e58a","observation_id":"7c42cf42-5dbb-4319-9999-5b6baa1e85c7","resolution":{"observed_at":"2026-05-09T00:24:28.374426Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-05-25T10:53:48.454413+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T10:53:48.454413+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00215","last_updated":"2024-06-28T19:53:17Z","snapshot_observed_at":"2026-07-06T18:38:46.314431Z","submitted_at":"2024-06-28T19:53:17Z","title":"LLM Critics Help Catch LLM Bugs","version":1},"cited_work":{"arxiv_id":"2407.00215","doi":"10.48550/arxiv.2407.00215","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.00215","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLM Critics Help Catch LLM Bugs","venue":"arXiv (Cornell University)","work_id":"e730c1aa-4c9b-48f3-923b-47ddc0a4e6d8","year":2024},"citing_paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-08T03:30:55.419225Z"},"links":{"cited_paper":"/paper/2407.00215","citing_paper":"/paper/2604.24955"},"observation_digest":"sha256:a79724796e20dcd25f6fe351741248e4819ad16da7aa51910e71bfcaf0c47223","observation_id":"954bd355-01ed-4240-a08c-369ca313ce8e","resolution":{"observed_at":"2026-05-09T00:24:28.369167Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.24955","last_updated":"2026-04-27T19:51:25Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T23:10:52.763251Z","submitted_at":"2026-04-27T19:51:25Z","title":"BenchGuard: Who Guards the Benchmarks? Automated Auditing of LLM Agent Benchmarks"},"reference_resolution":{"displayed":5,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":0,"verified_exact":3,"verified_fuzzy":0},"total_outbound_references":5},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 5 of 5 outbound references and 5 inbound Pith citation observations for arXiv:2604.24955."}