{"as_of":"2026-08-04T19:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a07d46b370b11ca218756168379531f389bb9931a165bd83bb3b69dc8f8c4b9f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":5,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":5,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T06:40:24.891449Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T12:15:01.137692Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":"2502.06781","doi":"10.48550/arxiv.2502.06781","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning","venue":null,"work_id":"48208898-31bc-4742-a673-7b6a4d77d027","year":2025},"citing_paper":{"arxiv_id":"2504.12501","last_updated":"2026-08-03T01:47:58Z","snapshot_observed_at":"2026-08-04T19:24:41.090409Z","submitted_at":"2025-04-16T21:36:46Z","title":"Reinforcement Learning from Human Feedback","version":9},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-22T19:27:40.991325Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2504.12501"},"observation_digest":"sha256:41697fc5ca2023641c64a8f1dddd0a90d3a0be0e5642a3d6bc5b53494fadff1a","observation_id":"40a1032b-0f5d-4e52-8a77-687f5ff1de13","resolution":{"observed_at":"2026-05-22T19:32:01.345899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-08-04T06:40:24.891449Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning.arXiv preprint arXiv:2502.06781, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.10739","last_updated":"2026-08-03T08:17:50Z","snapshot_observed_at":"2026-08-04T19:25:09.817430Z","submitted_at":"2025-12-11T15:26:28Z","title":"Intern-S1-MO: Long-horizon Reasoning Agent for Olympiad?Level Mathematical Problem Solving","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T06:40:24.891449Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2512.10739"},"observation_digest":"sha256:bd3f3c8b49255c02b2afa54c2c580c2c4eeeec7d9a7c88d137b2465217e42e06","observation_id":"da937472-cdc1-490a-beb9-4b79cd4fbc57","resolution":{"observed_at":"2026-08-04T06:40:24.891449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-08-03T03:04:43.981554Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning.arXiv preprint arXiv:2502.06781,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.09305","last_updated":"2026-06-28T18:13:37Z","snapshot_observed_at":"2026-08-03T07:10:48.024369Z","submitted_at":"2026-02-10T00:45:24Z","title":"Reward Modeling for Reinforcement Learning-Based LLM Reasoning: Design, Challenges, and Evaluation","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-03T03:04:43.981554Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2602.09305"},"observation_digest":"sha256:7771cf116d56c36a267967f2475d5836e204a075b4735c69887f30fa8249a360","observation_id":"cc15766a-f859-4eaa-9b3b-82060a08db9c","resolution":{"observed_at":"2026-08-03T03:04:43.981554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":"2502.06781","doi":"10.48550/arxiv.2502.06781","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning","venue":null,"work_id":"48208898-31bc-4742-a673-7b6a4d77d027","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":224,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:c36272ee177c0d44a77c9a62de5402772386400f0a27759a0bb4558d9372cf35","observation_id":"6e7d4cec-4529-43b9-bac8-67b44a0416a9","resolution":{"observed_at":"2026-07-01T20:56:13.534551Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":"2502.06781","doi":"10.48550/arxiv.2502.06781","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06781","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Exploring the limit of outcome reward for learning mathematical reasoning","venue":null,"work_id":"48208898-31bc-4742-a673-7b6a4d77d027","year":2025},"citing_paper":{"arxiv_id":"2606.11470","last_updated":"2026-06-09T21:59:37Z","snapshot_observed_at":"2026-07-06T23:50:35.052764Z","submitted_at":"2026-06-09T21:59:37Z","title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","version":1},"reference_index":165,"source":"arxiv_source","source_observed_at":"2026-06-27T12:59:51.091008Z"},"links":{"cited_paper":"/paper/2502.06781","citing_paper":"/paper/2606.11470"},"observation_digest":"sha256:c2954c6216be272060860185e5e59d763a411dfc60a42511bb265b0144a50ee3","observation_id":"7bc710db-20a8-4c95-a0e0-c252eb3cb39f","resolution":{"observed_at":"2026-06-27T13:00:56.109782Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.06781/citation-record","integrity":"/paper/2502.06781/integrity","json":"/paper/2502.06781/citation-record.json","paper":"/paper/2502.06781"},"outbound":[],"paper":{"arxiv_id":"2502.06781","last_updated":"2025-02-10T18:57:29Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T20:34:11.407726Z","submitted_at":"2025-02-10T18:57:29Z","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 5 inbound Pith citation observations for arXiv:2502.06781."}