{"as_of":"2026-08-05T03:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:780a0350e6cba93bbceb64a5228bf0127d26226fe01898cd9504579d38bd3182","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T18:52:52.969408Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-14T07:36:47.336670Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.02909","snapshot_observed_at":"2026-07-14T07:36:47.336670Z","title":"Dorner, Robin Staab, and Martin Vechev","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11022","last_updated":"2026-07-13T02:41:16Z","snapshot_observed_at":"2026-07-16T23:19:09.091393Z","submitted_at":"2026-07-13T02:41:16Z","title":"When the Reward Suite Is Leaky: A Preregistered Causal Contrast of Natural Verifier False Positives in RLVR","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-14T07:36:47.336670Z"},"links":{"cited_paper":"/paper/2605.02909","citing_paper":"/paper/2607.11022"},"observation_digest":"sha256:712e9ba30d77423b13694c765a3c67dfa22065613df4f18bfbd2c63bc8b24e94","observation_id":"59408003-0a79-4dac-9265-e33eae1e212f","resolution":{"observed_at":"2026-07-14T07:36:47.336670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2605.02909/citation-record","integrity":"/paper/2605.02909/integrity","json":"/paper/2605.02909/citation-record.json","paper":"/paper/2605.02909"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2508.06471","last_updated":"2025-08-08T17:21:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-08T17:21:06Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","version":1},"cited_work":{"arxiv_id":"2508.06471","doi":"10.48550/arxiv.2508.06471","metadata_source":"pith","pith_arxiv_id":"2508.06471","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","venue":"cs.CL","work_id":"5bb4e5d7-985e-431b-bb0d-75576cdc2950","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2508.06471","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:3fe3869155eb2c2f882fbbfbe4f3a1c5c4bc2ee28e156369062b2a7951eb5e9a","observation_id":"76519476-d543-4cbf-a667-29ebfd6c96cb","resolution":{"observed_at":"2026-05-11T17:50:08.654972Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T05:23:01.474969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T05:23:01.474969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:1310b77f130ceb23cc854638126fa262c439719e37e8d3384ead87744ba42df5","observation_id":"125e9404-bc66-4dca-a54f-220ea5b242bb","resolution":{"observed_at":"2026-05-10T23:45:53.096134Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"cited_work":{"arxiv_id":"2507.17746","doi":"10.48550/arxiv.2507.17746","metadata_source":"pith","pith_arxiv_id":"2507.17746","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","venue":"cs.LG","work_id":"805a846c-dae9-4375-abd8-a86dc6934496","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2507.17746","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:cd743fffda70a2bcb58d2b82bd195ee78e4791239296936e51c105d906a8b872","observation_id":"3b43911a-7a37-4e8b-b14f-9467ebd4b44b","resolution":{"observed_at":"2026-05-13T06:07:56.884714Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T21:53:02.879284+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T21:53:02.879284+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.04411","doi":"10.48550/arxiv.2601.04411","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Rate or fate? rlv r: Reinforcement learning with verifiable noisy rewards","venue":"arXiv (Cornell University)","work_id":"a97bd8c0-0af9-40f1-ac54-e7b6759bf1f4","year":2026},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:bc9cc212f9c49b15c0f35d3dd1730abb6498e4c55475b3006f609ade1c2e785a","observation_id":"5e6f45d2-ab36-4b79-8384-7d4ef0c296ad","resolution":{"observed_at":"2026-05-10T23:45:53.041417Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.00915","last_updated":"2026-05-22T12:16:11Z","snapshot_observed_at":"2026-08-02T22:13:21.681005Z","submitted_at":"2025-10-01T13:56:44Z","title":"Reinforcement Learning with Verifiable yet Noisy Rewards under Imperfect Verifiers","version":4},"cited_work":{"arxiv_id":"2510.00915","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.00915","snapshot_observed_at":"2026-07-04T17:09:58.372131Z","title":"Reinforcement Learning with Verifiable yet Noisy Rewards under Imperfect Verifiers","venue":"cs.LG","work_id":"12cfcad6-e768-4ecf-9455-0e5bdb645f9a","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2510.00915","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:6a68e882e8c3cc099a103f7cfd4ef0b674772bcf7dbf80138c1c7478de746ff7","observation_id":"8e146cf6-da76-484f-8e82-84205622ea8f","resolution":{"observed_at":"2026-05-25T03:01:06.910934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.22653","last_updated":"2025-05-28T17:59:03Z","snapshot_observed_at":"2026-07-06T21:32:23.537939Z","submitted_at":"2025-05-28T17:59:03Z","title":"The Climb Carves Wisdom Deeper Than the Summit: On the Noisy Rewards in Learning to Reason","version":1},"cited_work":{"arxiv_id":"2505.22653","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.22653","snapshot_observed_at":"2026-07-02T06:06:40.819064Z","title":"The climb carves wisdom deeper than the summit: On the noisy rewards in learning to reason","venue":null,"work_id":"6e345e0c-7c39-4e94-af60-56d7643c5704","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.22653","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:f602f4d8171042eff50065b34be0e32c0ce290078c10e3a1bed590958656e83c","observation_id":"108954cf-c305-4ec2-8327-6849d809a05d","resolution":{"observed_at":"2026-05-10T23:45:53.070918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.22203","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T06:06:40.751284Z","title":"Pitfalls of rule- and model-based verifiers–a case study on mathematical reasoning","venue":null,"work_id":"606127e0-cfd7-4548-893d-87fb5c3ea69e","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:bf0c6d6ccf760c0697e13cdc85931a1639e797b4acea4ad324c8f5450213b9c3","observation_id":"fada4551-957b-4858-8317-bc22f9733c1f","resolution":{"observed_at":"2026-05-10T23:45:52.968599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16400","last_updated":"2025-06-05T17:59:12Z","snapshot_observed_at":"2026-07-06T21:28:24.644692Z","submitted_at":"2025-05-22T08:50:47Z","title":"AceReason-Nemotron: Advancing Math and Code Reasoning through Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.16400","doi":"10.48550/arxiv.2505.16400","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16400","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Acereason-nemotron: Advancing math and code reasoning through reinforcement learning","venue":"ArXiv.org","work_id":"428ad314-c120-41da-9db7-b8bc1918fffb","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.16400","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:f145e0086c994846c5ebfec362653a04fb6690f25f6baaad497c4e6e5697e96d","observation_id":"edc3109a-db52-4c8f-b539-c362704567e8","resolution":{"observed_at":"2026-05-10T23:45:53.101099Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.08794","last_updated":"2026-06-11T04:21:36Z","snapshot_observed_at":"2026-07-06T21:55:50.206625Z","submitted_at":"2025-07-11T17:55:22Z","title":"One Token to Fool LLM-as-a-Judge","version":3},"cited_work":{"arxiv_id":"2507.08794","doi":"10.48550/arxiv.2507.08794","metadata_source":"pith","pith_arxiv_id":"2507.08794","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"One token to fool llm-as-a-judge","venue":"cs.LG","work_id":"f77305eb-5f89-4afb-84f3-51d864f3f3fa","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2507.08794","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:4483b3fdcd852190f8245abb8bfddd439986060b9d2f4212d05ac819e1903f07","observation_id":"12f775e4-d555-420d-91f6-8801f592afdd","resolution":{"observed_at":"2026-06-12T02:08:19.458599Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.24760","doi":"10.48550/arxiv.2505.24760","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reasoning gym: Reasoning environments for reinforcement learning with verifiable rewards.arXiv preprint arXiv:2505.24760","venue":"ArXiv.org","work_id":"dfc22a2c-1b8e-4318-a044-e142c0608571","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:694ea8671cd5d5dc6a4bf62bb51230c7e73f35e9a84f5538479832f1c0589eea","observation_id":"310b2449-9f46-43bb-87e0-94e585c3d0f2","resolution":{"observed_at":"2026-05-10T23:45:53.036235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:e7482b464e034489910e34fa0c24309a3240ca91dec2013d3b112172b47a62b7","observation_id":"694ad195-4ca0-4d2e-a90c-7b25eaa354b6","resolution":{"observed_at":"2026-05-10T23:45:53.128053Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"cited_work":{"arxiv_id":"2507.20534","doi":"10.1145/3448609","metadata_source":"pith","pith_arxiv_id":"2507.20534","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2: Open Agentic Intelligence","venue":"cs.LG","work_id":"7f18284c-12d3-4137-bea1-1da97e8cf3c1","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2507.20534","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:2bbc17751fd951187649cf1e84357eba6b6cb5637dd634fbd1105fe95cb37ad2","observation_id":"302b7624-4922-40a2-a517-468eae920624","resolution":{"observed_at":"2026-05-10T23:45:53.056780Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T01:23:16.170083+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:577a09e61c25ba30de8fd7fe2d17de2b32cf3d6430f9f733103d74b81c5dbdc5","observation_id":"c5ef677f-14c1-4d1a-aa23-1eb46de404c0","resolution":{"observed_at":"2026-05-10T23:45:52.995730Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.03613","last_updated":"2025-08-05T16:28:22Z","snapshot_observed_at":"2026-08-02T23:39:36.635461Z","submitted_at":"2025-08-05T16:28:22Z","title":"Goedel-Prover-V2: Scaling Formal Theorem Proving with Scaffolded Data Synthesis and Self-Correction","version":1},"cited_work":{"arxiv_id":"2508.03613","doi":"10.48550/arxiv.2508.03613","metadata_source":"pith","pith_arxiv_id":"2508.03613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Goedel-Prover-V2: Scaling Formal Theorem Proving with Scaffolded Data Synthesis and Self-Correction","venue":"cs.LG","work_id":"cfdb69bb-3c0b-41d3-bd34-3167a6931bb2","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2508.03613","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:53160ee7b37b39ab24990df9acc4bb286e26d19227dba48e43db38f76df192cc","observation_id":"d640b613-6f87-48a8-9aca-21fcbc27a503","resolution":{"observed_at":"2026-05-21T06:53:10.886894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-12T04:49:11.143969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T04:49:11.143969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"RLTF: reinforcement learning from unit test feedback","venue":null,"work_id":"7c0127cf-c687-4a6d-8643-e00e5d5ffe2b","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:a6827c5e4cf50007df81b5442d4e974b8917f268b2224c00dc4ad824366283ad","observation_id":"0406a932-dfb8-4e0c-a147-923635c8482c","resolution":{"observed_at":"2026-05-16T13:22:55.861396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18449","last_updated":"2025-12-01T00:16:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-25T18:45:04Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","version":2},"cited_work":{"arxiv_id":"2502.18449","doi":"10.48550/arxiv.2502.18449","metadata_source":"pith","pith_arxiv_id":"2502.18449","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","venue":"cs.SE","work_id":"4b93fb93-87c9-40fe-84d0-d7ecb4e11bed","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2502.18449","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:a4b18db297fed0ef65ebb5489ead1e6f3d7b904b3ea225b4405e57007694d293","observation_id":"1af04395-3132-40a2-90c1-4994093e95d2","resolution":{"observed_at":"2026-05-15T10:27:56.991478Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:f9accedbfea7442a650861dc265db9f8785c4d7201956f6e1adfba63b02d1465","observation_id":"f2a41475-0f73-4145-9baa-caaa43c43256","resolution":{"observed_at":"2026-05-10T23:45:52.981204Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20783","last_updated":"2025-10-06T09:30:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-26T17:59:14Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","version":2},"cited_work":{"arxiv_id":"2503.20783","doi":"10.48550/arxiv.2503.20783","metadata_source":"pith","pith_arxiv_id":"2503.20783","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding R1-Zero-Like Training: A Critical Perspective","venue":"cs.LG","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2503.20783","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:52583dae791b2ccb0a2be139bab8e73cdc1d12ac4a703017337fd160835ff62c","observation_id":"5ccaaeda-4316-4973-b9f5-36a828d9efa7","resolution":{"observed_at":"2026-05-10T23:45:52.974750Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:05.84445+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.20347","last_updated":"2025-12-01T12:02:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T14:25:19Z","title":"Soft Adaptive Policy Optimization","version":2},"cited_work":{"arxiv_id":"2511.20347","doi":"10.48550/arxiv.2511.20347","metadata_source":"pith","pith_arxiv_id":"2511.20347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Soft Adaptive Policy Optimization","venue":"cs.LG","work_id":"2d2c49f8-7feb-48c2-af7e-025b94c5d4f9","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2511.20347","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:c0c9728254616b601253bc8ed39c9b3165939b50e578b69fc27b7caa33d95941","observation_id":"893a8615-78e6-4c5a-84c2-66ad6ed087ac","resolution":{"observed_at":"2026-05-15T07:14:32.897338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is your code generated by chatgpt really correct? rigorous evaluation of large language models for code generation","venue":null,"work_id":"2b7e051f-7a24-452d-b3bb-0064170449bc","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:1e784dba7640f71ddb656758ccbe470064481735f121022072db1d3b5e93250e","observation_id":"43a860c9-3c65-400b-85f4-4d3c0c6a5a5e","resolution":{"observed_at":"2026-05-16T13:22:55.866308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14625","last_updated":"2025-05-22T17:49:50Z","snapshot_observed_at":"2026-08-03T22:59:07.639277Z","submitted_at":"2025-05-20T17:16:44Z","title":"TinyV: Reducing False Negatives in Verification Improves RL for LLM Reasoning","version":2},"cited_work":{"arxiv_id":"2505.14625","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.14625","snapshot_observed_at":"2026-07-02T02:16:26.183764Z","title":"Tinyv: Reducing false negatives in verification improves rl for llm reasoning","venue":null,"work_id":"851de516-697c-423c-8c2b-4f5fbdf5ac13","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2505.14625","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:13f35f90eddb78c8a661ea090145ee234d0f3d97725267b0145b913879ba524d","observation_id":"42f34f68-8adb-4ba9-baf8-69569e51f279","resolution":{"observed_at":"2026-05-10T23:45:53.117674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Is llm-as-a-judge robust? investigating universal adversarial attacks on zero-shot llm assessment","venue":null,"work_id":"138ade5b-7e26-4ede-bb04-5f663b09fd96","year":2024},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:8ba586f830f0f0675fec5546a12f8d6d3d9fb91ba84e1cee842d651177c71479","observation_id":"710d6935-db0d-49c8-9d07-d62b7b60cb34","resolution":{"observed_at":"2026-05-16T13:22:55.876070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17995","last_updated":"2026-04-14T03:25:43Z","snapshot_observed_at":"2026-08-02T10:58:14.984747Z","submitted_at":"2025-09-22T16:36:56Z","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","version":2},"cited_work":{"arxiv_id":"2509.17995","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.17995","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Variation in Verification: Understanding Verification Dynamics in Large Language Models","venue":"cs.CL","work_id":"f614fdf8-06a5-45a4-a7b0-d2ded78a7289","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2509.17995","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:07537d9a4142249bbec9844bf43c6c171d1c97d4014c88dd5bcea2b09e51d7f2","observation_id":"f37d6420-c18b-4d4b-98c2-91da1bae9668","resolution":{"observed_at":"2026-05-10T23:45:53.052014Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Verifybench: A systematic benchmark for evaluating reasoning verifiers across domains","venue":null,"work_id":"6b608204-4555-4335-8e2d-d9d5cfa98293","year":2026},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:d32c13cba8e3a922d64c6e179da329604b663a14ece5ffda8c8960cc5eed6222","observation_id":"73724eae-10b4-4a4f-a4a0-3e9924140e7e","resolution":{"observed_at":"2026-05-16T13:22:55.868976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.15801","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pure Python","venue":null,"work_id":"65a36a53-2af5-4e5a-82ff-806809a22ae4","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:e68ad1ffc4276d6e20a05e4ff2667414f03df6a573d8b4593208ca60ca73fd6d","observation_id":"c42e22a1-a308-4f1a-976e-c9143320e6d0","resolution":{"observed_at":"2026-05-10T23:45:53.008410Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.16140","last_updated":"2026-07-29T22:29:54Z","snapshot_observed_at":"2026-08-02T23:16:34.332808Z","submitted_at":"2026-03-17T05:48:32Z","title":"Noisy Data is Destructive to Reinforcement Learning with Verifiable Rewards","version":2},"cited_work":{"arxiv_id":"2603.16140","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2603.16140","snapshot_observed_at":"2026-07-31T02:03:17.773516Z","title":"Noisy data is destructive to reinforcement learning with verifiable rewards","venue":null,"work_id":"b0966eb6-27a5-4d71-9205-f8df1710d940","year":2026},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2603.16140","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:63325602d6e5b2a6f2b6c55dc2ff9f237ff2c3b54c41632530f28b8c371218cf","observation_id":"fc0d0878-2e23-48f6-add8-555fa72e8844","resolution":{"observed_at":"2026-07-31T02:03:17.773516Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Russell, and Anca D","venue":null,"work_id":"a4606a2d-cb02-4892-836f-30997132bfc4","year":2017},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:2b2bc5c49ec564ba76093ec9b93244243d0677ee262ccd25e8db9940103e4485","observation_id":"c9795332-5524-49f9-8460-35aa937a5903","resolution":{"observed_at":"2026-05-16T13:22:55.855833Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.13085","last_updated":"2025-03-05T21:08:30Z","snapshot_observed_at":"2026-07-06T13:56:34.336341Z","submitted_at":"2022-09-27T00:32:44Z","title":"Defining and Characterizing Reward Hacking","version":2},"cited_work":{"arxiv_id":"2209.13085","doi":"10.48550/arxiv.2209.13085","metadata_source":"pith","pith_arxiv_id":"2209.13085","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Skalse, N","venue":"cs.LG","work_id":"b9869329-2719-4e51-889d-b637ee5e468d","year":2022},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2209.13085","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:5c5964f971f1c2ae2800663e4a35ceb0e9b9435a85a0795dca7c60ab18a7f127","observation_id":"bfb8015f-5e33-47bc-999f-1209b38f7bdf","resolution":{"observed_at":"2026-05-10T19:00:45.572142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scaling laws for reward model overoptimization","venue":null,"work_id":"19dca190-8e4f-4d48-8365-3745a7f65cfe","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:b660d352cf4d1d1e9113e1059d897b898c16dba191c94927baab10bb90dc4aee","observation_id":"2101020d-97fa-4020-9d94-224294c23fdf","resolution":{"observed_at":"2026-05-16T13:22:55.863900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.12399","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Roc-n-reroll: How verifier imperfection affects test-time scaling","venue":null,"work_id":"6647d287-2088-4c7a-b442-37a3621ea702","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:9029ffd3d1ff8d7e739b9049c4de0b8001c4a2707a439f1cc30b512ef839bed1","observation_id":"2e819090-0377-429c-8f40-bf42889e2576","resolution":{"observed_at":"2026-05-10T23:45:52.989245Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T10:53:40.936134Z","title":"TRL: Transformers Reinforcement Learning","venue":null,"work_id":"6e7d325d-c920-4461-9919-8108e384812e","year":2020},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:de024638fd5441da5df6b44c5ac022529781487230bccccf0024252b09f2bef7","observation_id":"425e4519-bee4-4879-9351-fc61e21d47a4","resolution":{"observed_at":"2026-05-16T13:22:55.858818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13961","last_updated":"2026-04-14T15:12:44Z","snapshot_observed_at":"2026-07-06T22:39:08.850289Z","submitted_at":"2025-12-15T23:41:48Z","title":"Olmo 3","version":2},"cited_work":{"arxiv_id":"2512.13961","doi":"10.1007/s13755-026-00428-z","metadata_source":"pith","pith_arxiv_id":"2512.13961","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Olmo 3","venue":"cs.CL","work_id":"74de5f5e-0a69-4f73-862d-e5705fa9f4bb","year":2025},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"cited_paper":"/paper/2512.13961","citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:3fa304abe4d73a17b8f5d469a2c6c47f3d49f7b554a1dd0e0e9e637d1e8eaabe","observation_id":"33d62132-cfd7-49b1-8b6b-a3cc2e0494b7","resolution":{"observed_at":"2026-05-10T23:45:53.027215Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"langdetect","venue":null,"work_id":"216ed148-aab5-43f6-8c62-040dbdf5a5d5","year":2021},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:91179945c28f7dd2cb48b711dba6d46ef639704ebe6a24e4b7c609774d2ae49a","observation_id":"6fc3895d-9e6b-4ee6-a934-c175b8e1e015","resolution":{"observed_at":"2026-05-16T13:22:55.871163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":"52a458b7-066f-4f6a-87b2-1bfa236cde69","year":2023},"citing_paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-10T18:52:52.969408Z"},"links":{"citing_paper":"/paper/2605.02909"},"observation_digest":"sha256:ea1168d8afc8589c97cd5f87c55deb1a9e021cea811a3d3619a6f2cb1248d6e1","observation_id":"44a3a9e5-c032-45a3-99be-5a4096b36879","resolution":{"observed_at":"2026-05-16T13:22:55.873708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.02909","last_updated":"2026-04-06T15:02:52Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-06T15:02:52Z","title":"Delay, Plateau, or Collapse: Evaluating the Impact of Systematic Verification Error on RLVR"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":24,"verified_fuzzy":9},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 1 inbound Pith citation observation for arXiv:2605.02909."}