{"as_of":"2026-08-07T04:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:986388b8ef256732e31ddc594c031408df742ddc7de06113b6525c320a8d0f99","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:41:48.404875Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-07T00:41:48.404875Z","title":"Xiong, H","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12928","last_updated":"2025-06-15T17:59:47Z","snapshot_observed_at":"2026-08-07T00:35:06.984805Z","submitted_at":"2025-06-15T17:59:47Z","title":"Scaling Test-time Compute for LLM Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:41:48.404875Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2506.12928"},"observation_digest":"sha256:64391ad8cc92bde80bb5ef78a4339456c78b999a10ef5f3875603e33066a68da","observation_id":"97f62b28-d998-499f-aa58-9eb0923c1a72","resolution":{"observed_at":"2026-08-07T00:41:48.404875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2509.03403","last_updated":"2026-05-15T21:29:46Z","snapshot_observed_at":"2026-07-06T22:23:00.681940Z","submitted_at":"2025-09-03T15:28:51Z","title":"Beyond Correctness: Harmonizing Process and Outcome Rewards through RL Training","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-21T22:38:57.833414Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2509.03403"},"observation_digest":"sha256:b67196c703982d0a77bc14cf020926abd1132c22da1ef6445a9d11fd71fb0b6f","observation_id":"cc1bd704-6917-4eda-a7cf-32a282e1a413","resolution":{"observed_at":"2026-05-21T22:40:43.171530Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2510.07962","last_updated":"2026-05-21T01:02:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-09T08:55:12Z","title":"LightReasoner: Can Small Language Models Teach Large Language Models Reasoning?","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T12:59:31.011283Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2510.07962"},"observation_digest":"sha256:550d49e11e5e313624e7924a7bdda07ff5216ee2feee919668027aa2dd107ac8","observation_id":"06743f5c-b56f-4ab0-a5ae-fcede9ccce86","resolution":{"observed_at":"2026-05-22T13:01:34.019544Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-04T10:44:31.460549Z","title":"Self-rewarding correction for mathematical reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.08977","last_updated":"2026-06-02T08:25:17Z","snapshot_observed_at":"2026-08-06T09:42:32.578654Z","submitted_at":"2025-10-10T03:38:17Z","title":"Breaking the Self-Confirming Loop: Diagnosing and Mitigating Systemic Reward Bias in Self-Rewarding RL","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T10:44:31.460549Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2510.08977"},"observation_digest":"sha256:447b4806ebbc50ec0f57382ab4b2113c268e5ffdc12c7101eabdeea733277652","observation_id":"5830c2c4-5a28-44d0-9fea-33295906121a","resolution":{"observed_at":"2026-08-04T10:44:31.460549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-03T05:14:21.533431Z","title":"Self-rewarding correction for mathematical reasoning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.02979","last_updated":"2026-05-24T21:03:36Z","snapshot_observed_at":"2026-08-03T05:14:15.579080Z","submitted_at":"2026-02-03T01:38:53Z","title":"CPMobius: Iterative Coach-Player Reasoning for Data-Free Reinforcement Learning","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-03T05:14:21.533431Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2602.02979"},"observation_digest":"sha256:70e1327458fd98072751e3f6b8437c033f36c27aaa64e9b044e1abb83e06b0ad","observation_id":"b896a909-8106-4d06-ac64-1436ef0624b9","resolution":{"observed_at":"2026-08-03T05:14:21.533431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-03T03:33:41.469348Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-21T13:13:13.293921Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:e41ab39e1d81411c365c37d9f29a4d93de2fd38168a75083a5bbd22ace9c0c61","observation_id":"81f621f4-3604-428e-8cc2-e0bf2c72abc3","resolution":{"observed_at":"2026-05-21T13:14:10.976923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-03T03:33:45.150810Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.07832","last_updated":"2026-07-03T17:35:46Z","snapshot_observed_at":"2026-08-03T03:33:41.469348Z","submitted_at":"2026-02-08T05:47:27Z","title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T03:33:45.150810Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2602.07832"},"observation_digest":"sha256:c6e6edbd7cfbf60526b7105b39d8c109880dba41a320ceab06da0be0d0442da9","observation_id":"dbfc3b72-9643-43e4-85a6-9b411c13cfd7","resolution":{"observed_at":"2026-08-03T03:33:45.150810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2604.03993","last_updated":"2026-04-05T06:30:50Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T06:30:50Z","title":"Can LLMs Learn to Reason Robustly under Noisy Supervision?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T16:58:42.129870Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2604.03993"},"observation_digest":"sha256:ca703c043647a7a8a91cb143f0726238caefa2331d250a9104670d9d3dafad40","observation_id":"45e07ccf-0df1-4722-b143-7e48f81aada3","resolution":{"observed_at":"2026-05-13T17:08:01.264453Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.05226","last_updated":"2026-05-23T13:20:07Z","snapshot_observed_at":"2026-07-06T23:17:52.987860Z","submitted_at":"2026-04-19T10:33:19Z","title":"Internalizing Outcome Supervision into Process Supervision: A New Paradigm for Reinforcement Learning for Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T06:13:09.898530Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.05226"},"observation_digest":"sha256:f1e84c4e27a478d0c88110868e0646e212fbffff5aef52a3f1bab5b92ce25df2","observation_id":"6c7cb4b3-0047-45bd-a6ed-f07c5cd35174","resolution":{"observed_at":"2026-05-10T06:26:27.829461Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.16299","last_updated":"2026-05-21T14:01:21Z","snapshot_observed_at":"2026-08-02T19:56:28.579984Z","submitted_at":"2026-04-17T07:20:01Z","title":"ACE: Self-Evolving LLM Coding Framework via Adversarial Unit Test Generation and Preference Optimization","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T00:55:40.784295Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.16299"},"observation_digest":"sha256:51f7413d3f2766e175bed1af75f8f132e66a38aae90ffd51ad99ae433f6cf69b","observation_id":"cf61d227-58ea-4435-9385-6261c2420a7c","resolution":{"observed_at":"2026-05-21T00:59:19.336113Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.16299","last_updated":"2026-05-21T14:01:21Z","snapshot_observed_at":"2026-08-02T19:56:28.579984Z","submitted_at":"2026-04-17T07:20:01Z","title":"ACE: Self-Evolving LLM Coding Framework via Adversarial Unit Test Generation and Preference Optimization","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T10:14:00.478000Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.16299"},"observation_digest":"sha256:a1557b5601a0da5f60596a59f64aa209e8569ae159b2a349e795c772dda2d16f","observation_id":"f414539e-896d-4e51-bb05-6e65dc61007e","resolution":{"observed_at":"2026-05-22T10:14:47.165809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2605.24613","last_updated":"2026-05-23T14:51:13Z","snapshot_observed_at":"2026-07-06T23:34:43.210149Z","submitted_at":"2026-05-23T14:51:13Z","title":"Guarded Repair for Harm-Aware Post-hoc Replacement of LLM Mathematical Reasoning","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-30T13:38:48.617204Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2605.24613"},"observation_digest":"sha256:f889526ba2a119e18b80df14bc995548ea903af06ec6e831920224ff7d6923a7","observation_id":"0dd7aacd-e04f-4881-a417-eea4ab73c190","resolution":{"observed_at":"2026-06-30T13:44:40.809593Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":162,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:5ef91c568ff9c8652aa179c39db7bbe54235c24f90a5404514d69776d3b9fd5c","observation_id":"03fe794f-2c38-44d0-961d-2ae2f83409a2","resolution":{"observed_at":"2026-07-01T20:56:13.435749Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.11470","last_updated":"2026-06-09T21:59:37Z","snapshot_observed_at":"2026-07-06T23:50:35.052764Z","submitted_at":"2026-06-09T21:59:37Z","title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","version":1},"reference_index":274,"source":"arxiv_source","source_observed_at":"2026-06-27T12:59:51.091008Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.11470"},"observation_digest":"sha256:ad8d7893eb836218b61c72c882de86f1d75bebe890dadcacaa6814975b3c7057","observation_id":"c9f7a4fa-0632-4437-b449-b34408d22a14","resolution":{"observed_at":"2026-06-27T13:00:55.970436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.13316","last_updated":"2026-07-31T08:57:20Z","snapshot_observed_at":"2026-08-05T23:10:47.877661Z","submitted_at":"2026-06-11T13:10:48Z","title":"ReSum: Synergizing LLM Reasoning and Summarization with Reinforcement Learning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T06:39:34.199607Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.13316"},"observation_digest":"sha256:f1e41a920b1bfe089d0d86ce4b4940f02f53ef18c397c4d050f3ed93ba191995","observation_id":"86fff995-f14c-4448-b088-6cff9d7586e5","resolution":{"observed_at":"2026-07-03T15:08:33.388470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-03T02:12:24.010055Z","title":"Xiong, H","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.13316","last_updated":"2026-07-31T08:57:20Z","snapshot_observed_at":"2026-08-05T23:10:47.877661Z","submitted_at":"2026-06-11T13:10:48Z","title":"ReSum: Synergizing LLM Reasoning and Summarization with Reinforcement Learning","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-03T02:12:24.010055Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.13316"},"observation_digest":"sha256:2ddaa357144636f8324b062d1805a1b98f2f15693f01d795bf759c245aeda720","observation_id":"deeadaf6-7404-490b-a02c-e0291d1527f5","resolution":{"observed_at":"2026-08-03T02:12:24.010055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.21943","last_updated":"2026-06-20T08:20:41Z","snapshot_observed_at":"2026-07-06T23:56:54.959593Z","submitted_at":"2026-06-20T08:20:41Z","title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","version":1},"reference_index":238,"source":"pdf_text","source_observed_at":"2026-06-26T12:15:08.304150Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.21943"},"observation_digest":"sha256:7acafa6bd8da1358c1b5928c1d83ad5ca4603cbda201f8f76c21ef2e5582682a","observation_id":"2cf7f342-4619-40d3-8df7-e5cc461b49d7","resolution":{"observed_at":"2026-07-04T07:59:40.132829Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":"2502.19613","doi":"10.48550/arxiv.2502.19613","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613","venue":"ArXiv.org","work_id":"1ceb5bfe-8d1e-431e-b67a-458ea30e168e","year":2025},"citing_paper":{"arxiv_id":"2606.28186","last_updated":"2026-07-11T22:18:18Z","snapshot_observed_at":"2026-07-16T23:17:54.586391Z","submitted_at":"2026-06-26T15:32:17Z","title":"Cognitive Episodes in LLM Reasoning Traces Enable Interpretable Human Item Difficulty Prediction","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T03:58:32.372896Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.28186"},"observation_digest":"sha256:07f40471e575b3edb2a14bc5b79b69f16ce4257c1e67eff513f9aa7f61195050","observation_id":"af7e3537-efcf-4951-ba7a-d7583bc4b8f7","resolution":{"observed_at":"2026-07-01T17:15:51.588940Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19613","snapshot_observed_at":"2026-07-14T17:12:24.565155Z","title":"Self-rewarding correction for mathematical reasoning.arXiv preprint arXiv:2502.19613,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.28186","last_updated":"2026-07-11T22:18:18Z","snapshot_observed_at":"2026-07-16T23:17:54.586391Z","submitted_at":"2026-06-26T15:32:17Z","title":"Cognitive Episodes in LLM Reasoning Traces Enable Interpretable Human Item Difficulty Prediction","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-14T17:12:24.565155Z"},"links":{"cited_paper":"/paper/2502.19613","citing_paper":"/paper/2606.28186"},"observation_digest":"sha256:f46de7812adbec5ec4987bfb4c24e7c6e3e26eb9230c90a6b74fff06d292b28f","observation_id":"8428ec32-f496-4353-9265-0eb5a8b61781","resolution":{"observed_at":"2026-07-14T17:12:24.565155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.19613/citation-record","integrity":"/paper/2502.19613/integrity","json":"/paper/2502.19613/citation-record.json","paper":"/paper/2502.19613"},"outbound":[],"paper":{"arxiv_id":"2502.19613","last_updated":"2025-02-26T23:01:16Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T20:43:22.572496Z","submitted_at":"2025-02-26T23:01:16Z","title":"Self-rewarding correction for mathematical reasoning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2502.19613."}