{"as_of":"2026-08-08T14:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:825bbf8f8a2a64babf4a6dee636fde293910f80cde8afe84d56d5c9ca6092d82","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":9,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":9,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:27:26.145226Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-07T11:27:26.145226Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02480","last_updated":"2025-06-03T05:51:35Z","snapshot_observed_at":"2026-08-07T11:20:46.420557Z","submitted_at":"2025-06-03T05:51:35Z","title":"ORPP: Self-Optimizing Role-playing Prompts to Enhance Language Model Capabilities","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T11:27:26.145226Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2506.02480"},"observation_digest":"sha256:74ece65bb64dc3e6033b52a5fdb723372d568e8539c59c628e8923a964600ee6","observation_id":"a54cd0d1-3536-4eff-83f8-93f98202586d","resolution":{"observed_at":"2026-08-07T11:27:26.145226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-06T18:16:47.983056Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08924","last_updated":"2025-07-18T09:31:19Z","snapshot_observed_at":"2026-08-07T23:55:32.686302Z","submitted_at":"2025-07-11T17:56:32Z","title":"From KMMLU-Redux to KMMLU-Pro: A Professional Korean Benchmark Suite for LLM Evaluation","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T18:16:47.983056Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2507.08924"},"observation_digest":"sha256:e97ce1d11f614a536f5edfdb93290cb6463accbf60846295854cd29bdd735b3a","observation_id":"5e135f8a-94d1-4679-acea-e22753019b80","resolution":{"observed_at":"2026-08-06T18:16:47.983056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":"2412.15194","doi":"10.48550/arxiv.2412.15194","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmlu-cf: A contamination-free multi-task language understanding benchmark","venue":"arXiv (Cornell University)","work_id":"adfea584-cb25-40f7-8839-dc5e738e5fa0","year":2024},"citing_paper":{"arxiv_id":"2604.04942","last_updated":"2026-03-13T13:01:01Z","snapshot_observed_at":"2026-07-06T22:53:46.999911Z","submitted_at":"2026-03-13T13:01:01Z","title":"TDA-RC: Task-Driven Alignment for Knowledge-Based Reasoning Chains in Large Language Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-15T12:21:39.237267Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2604.04942"},"observation_digest":"sha256:391c8cd1180bc87e2dc7d243e53a3bc6484a565ed71f8e83619cd3962ad269c6","observation_id":"e689eac6-1808-4201-a231-73f69a312a71","resolution":{"observed_at":"2026-05-15T12:25:36.032153Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":"2412.15194","doi":"10.48550/arxiv.2412.15194","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmlu-cf: A contamination-free multi-task language understanding benchmark","venue":"arXiv (Cornell University)","work_id":"adfea584-cb25-40f7-8839-dc5e738e5fa0","year":2024},"citing_paper":{"arxiv_id":"2604.15972","last_updated":"2026-04-17T11:36:20Z","snapshot_observed_at":"2026-07-06T23:03:26.308324Z","submitted_at":"2026-04-17T11:36:20Z","title":"Weak-Link Optimization for Multi-Agent Reasoning and Collaboration","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T08:53:03.420926Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2604.15972"},"observation_digest":"sha256:614ae6bfe6e9f6da052ee6bfb2daec253f8d96a5e208087f8bbf97288189c1b4","observation_id":"309c7c76-6f33-4ca6-bc7e-32a15024d37e","resolution":{"observed_at":"2026-05-10T08:58:13.183812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":"2412.15194","doi":"10.48550/arxiv.2412.15194","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmlu-cf: A contamination-free multi-task language understanding benchmark","venue":"arXiv (Cornell University)","work_id":"adfea584-cb25-40f7-8839-dc5e738e5fa0","year":2024},"citing_paper":{"arxiv_id":"2605.21543","last_updated":"2026-05-20T09:16:39Z","snapshot_observed_at":"2026-08-01T19:04:50.092578Z","submitted_at":"2026-05-20T09:16:39Z","title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","version":1},"reference_index":180,"source":"arxiv_source","source_observed_at":"2026-05-22T00:40:54.038367Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2605.21543"},"observation_digest":"sha256:9d41b08e9985e91402a611439005943c95f4077511e241689dd99d18f3f099e6","observation_id":"3ee3fae2-f0b0-40c6-b5a5-b7954c930d0f","resolution":{"observed_at":"2026-05-22T00:44:29.471191Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":"2412.15194","doi":"10.48550/arxiv.2412.15194","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmlu-cf: A contamination-free multi-task language understanding benchmark","venue":"arXiv (Cornell University)","work_id":"adfea584-cb25-40f7-8839-dc5e738e5fa0","year":2024},"citing_paper":{"arxiv_id":"2606.26396","last_updated":"2026-06-24T21:26:43Z","snapshot_observed_at":"2026-08-02T16:34:16.438122Z","submitted_at":"2026-06-24T21:26:43Z","title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-06-26T01:27:39.812228Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2606.26396"},"observation_digest":"sha256:c1e17df294f3a03574cd0a6c2141acefd15fab3a4278a7e274d31a6348471837","observation_id":"c8b490c7-7a77-43b4-a1d2-0ef8d2636d42","resolution":{"observed_at":"2026-06-26T01:28:50.524986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-07-14T15:45:54.532529Z","title":"MMLU-CF : A contamination-free multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09786","last_updated":"2026-08-02T18:21:10Z","snapshot_observed_at":"2026-08-07T05:12:04.880594Z","submitted_at":"2026-07-08T14:18:26Z","title":"Length Penalties Make Chain-of-Thought Less Monitorable","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-07-14T15:45:54.532529Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2607.09786"},"observation_digest":"sha256:2bf578ca82d588308e227af8b8ec004634858bb0b0afb5a0b6377efdbc2cd963","observation_id":"cce7df17-337f-413b-8be4-643d567840e5","resolution":{"observed_at":"2026-07-14T15:45:54.532529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-02T08:06:10.871998Z","title":"MMLU-CF : A contamination-free multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09786","last_updated":"2026-08-02T18:21:10Z","snapshot_observed_at":"2026-08-07T05:12:04.880594Z","submitted_at":"2026-07-08T14:18:26Z","title":"Length Penalties Make Chain-of-Thought Less Monitorable","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-02T08:06:10.871998Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2607.09786"},"observation_digest":"sha256:e3b1ef81d2bf79610279652002ceb559bbf93a5e02292393c008fc5a04f4d6ae","observation_id":"e9e5c36a-ae8a-4a43-a059-dc7cc848603d","resolution":{"observed_at":"2026-08-02T08:06:10.871998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15194","snapshot_observed_at":"2026-08-04T04:30:30.646103Z","title":"MMLU-CF : A contamination-free multi-task language understanding benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.09786","last_updated":"2026-08-02T18:21:10Z","snapshot_observed_at":"2026-08-07T05:12:04.880594Z","submitted_at":"2026-07-08T14:18:26Z","title":"Length Penalties Make Chain-of-Thought Less Monitorable","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T04:30:30.646103Z"},"links":{"cited_paper":"/paper/2412.15194","citing_paper":"/paper/2607.09786"},"observation_digest":"sha256:eed314c0991697b982abb0470931957322e2ea13b1afbfd4c0ed83bf6d4ecb51","observation_id":"468f8bbb-9c8c-448c-accf-13bc3d193157","resolution":{"observed_at":"2026-08-04T04:30:30.646103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.15194/citation-record","integrity":"/paper/2412.15194/integrity","json":"/paper/2412.15194/citation-record.json","paper":"/paper/2412.15194"},"outbound":[],"paper":{"arxiv_id":"2412.15194","last_updated":"2024-12-19T18:58:04Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T20:10:24.217625Z","submitted_at":"2024-12-19T18:58:04Z","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 9 inbound Pith citation observations for arXiv:2412.15194."}