{"as_of":"2026-08-20T04:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:49b1395370dbfc64e2bc5fd1e885e7a1db3730b01fdb9fc98770e059c9012b3f","coverage":[{"denominator":3,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T07:32:32.771146Z","state":"measured"},{"denominator":4,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":4,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:58:40.067379Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-15T14:58:40.858059Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.30504","last_updated":"2026-05-28T19:38:44Z","snapshot_observed_at":"2026-08-16T13:13:25.583251Z","submitted_at":"2026-05-28T19:38:44Z","title":"Auditing LLM Benchmarks with Item Response Theory","version":1},"cited_work":{"arxiv_id":"2605.30504","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.30504","snapshot_observed_at":"2026-08-15T14:58:40.858059Z","title":"Auditing LLM Benchmarks with Item Response Theory","venue":"cs.CL","work_id":"bf22ec97-0390-4ce3-a2c0-bb5fc055e2e4","year":2026},"citing_paper":{"arxiv_id":"2608.02966","last_updated":"2026-08-03T23:59:39Z","snapshot_observed_at":"2026-08-18T05:28:10.742527Z","submitted_at":"2026-08-03T23:59:39Z","title":"Every Wrong Answer Counts: Option-Level Psychometrics for LLM Multiple-Choice Benchmarks","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-15T14:58:40.067379Z"},"links":{"cited_paper":"/paper/2605.30504","citing_paper":"/paper/2608.02966"},"observation_digest":"sha256:c53deceee5bddc0ce6aa0b9e26226a2c8b524c99a7667bbc7e3e61a333aebe46","observation_id":"9cfeaf8c-6499-47f7-a276-c9b42cfd7208","resolution":{"observed_at":"2026-08-15T14:58:40.949418Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.30504/citation-record","integrity":"/paper/2605.30504/integrity","json":"/paper/2605.30504/citation-record.json","paper":"/paper/2605.30504"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2511.04689","last_updated":"2026-07-13T22:56:21Z","snapshot_observed_at":"2026-08-18T18:20:27.842811Z","submitted_at":"2025-10-26T03:54:12Z","title":"Adaptive Testing for LLM Evaluation: A Psychometric Alternative to Static Benchmarks","version":3},"cited_work":{"arxiv_id":"2511.04689","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2511.04689","snapshot_observed_at":"2026-07-15T01:20:49.385427Z","title":"Frederic M","venue":null,"work_id":"84eb4ef4-b147-4b98-a271-127b0a97d297","year":2025},"citing_paper":{"arxiv_id":"2605.30504","last_updated":"2026-05-28T19:38:44Z","snapshot_observed_at":"2026-08-16T13:13:25.583251Z","submitted_at":"2026-05-28T19:38:44Z","title":"Auditing LLM Benchmarks with Item Response Theory","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T07:32:32.771146Z"},"links":{"cited_paper":"/paper/2511.04689","citing_paper":"/paper/2605.30504"},"observation_digest":"sha256:50c55fea5074da4f5e101c5315f9d1c6006c155951cdb922e12d0d4ea6cc168b","observation_id":"ce90103f-09e4-465b-b8e2-40f35029bfc6","resolution":{"observed_at":"2026-07-15T01:20:49.385427Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16789","last_updated":"2024-03-09T22:26:06Z","snapshot_observed_at":"2026-08-08T18:07:29.632928Z","submitted_at":"2023-10-25T17:21:23Z","title":"Detecting Pretraining Data from Large Language Models","version":3},"cited_work":{"arxiv_id":"2310.16789","doi":"10.48550/arxiv.2310.16789","metadata_source":"pith","pith_arxiv_id":"2310.16789","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Detecting Pretraining Data from Large Language Models","venue":"cs.CL","work_id":"1ff0530f-0b29-487b-ba43-d22a740293b1","year":2023},"citing_paper":{"arxiv_id":"2605.30504","last_updated":"2026-05-28T19:38:44Z","snapshot_observed_at":"2026-08-16T13:13:25.583251Z","submitted_at":"2026-05-28T19:38:44Z","title":"Auditing LLM Benchmarks with Item Response Theory","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-29T07:32:32.771146Z"},"links":{"cited_paper":"/paper/2310.16789","citing_paper":"/paper/2605.30504"},"observation_digest":"sha256:0d9842e24873ecfba4ea7c8bb1afe147dde91b64479c464ff1448615872e12ba","observation_id":"b4c8a0ec-6a87-4080-b871-934a928b56d1","resolution":{"observed_at":"2026-06-29T07:33:13.330613Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03461","last_updated":"2025-02-05T18:58:19Z","snapshot_observed_at":"2026-08-18T09:40:39.257250Z","submitted_at":"2025-02-05T18:58:19Z","title":"Do Large Language Model Benchmarks Test Reliability?","version":1},"cited_work":{"arxiv_id":"2502.03461","doi":"10.48550/arxiv.2502.03461","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.03461","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Do large language model benchmarks test reliability?","venue":"ArXiv.org","work_id":"11771fd3-b9c9-4098-a437-00e057657bb7","year":2025},"citing_paper":{"arxiv_id":"2605.30504","last_updated":"2026-05-28T19:38:44Z","snapshot_observed_at":"2026-08-16T13:13:25.583251Z","submitted_at":"2026-05-28T19:38:44Z","title":"Auditing LLM Benchmarks with Item Response Theory","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T07:32:32.771146Z"},"links":{"cited_paper":"/paper/2502.03461","citing_paper":"/paper/2605.30504"},"observation_digest":"sha256:127a859ad375b6712d3d2928c6b4be5a4cf6b4c4a49af3c740b42bf3290633fc","observation_id":"7f005ee2-3673-4666-9e11-b68055fe34a7","resolution":{"observed_at":"2026-06-29T07:33:13.327707Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.30504","last_updated":"2026-05-28T19:38:44Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T13:13:25.583251Z","submitted_at":"2026-05-28T19:38:44Z","title":"Auditing LLM Benchmarks with Item Response Theory"},"reference_resolution":{"displayed":3,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":3},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 3 of 3 outbound references and 1 inbound Pith citation observation for arXiv:2605.30504."}