{"as_of":"2026-08-07T10:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:aca5f16e7a98ef6dd0e064c54e0477a8fd311c95d78c5cf2ab3e37c168568688","coverage":[{"denominator":18,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T07:33:55.284758Z","state":"measured"},{"denominator":18,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":18,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.29296/citation-record","integrity":"/paper/2606.29296/integrity","json":"/paper/2606.29296/citation-record.json","paper":"/paper/2606.29296"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:b25e09aca042fed1bab43fc12d9fe11bf31ae11ce679bdf40ef0d5e61c8ea061","observation_id":"4bd02a22-1010-4a61-a1af-0e9881eb6641","resolution":{"observed_at":"2026-06-30T07:34:21.180773Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:441ddfef812ba78393a1339eeb7de9c0f355b3ed57b6d65c6d40bcfdb2a1327e","observation_id":"1cdd8a9e-171f-450b-9f4f-b7a2e0a06c6c","resolution":{"observed_at":"2026-06-30T07:34:20.679568Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.03403","last_updated":"2026-05-15T21:29:46Z","snapshot_observed_at":"2026-07-06T22:23:00.681940Z","submitted_at":"2025-09-03T15:28:51Z","title":"Beyond Correctness: Harmonizing Process and Outcome Rewards through RL Training","version":2},"cited_work":{"arxiv_id":"2509.03403","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.03403","snapshot_observed_at":"2026-07-03T16:48:39.972038Z","title":"14 Preprint","venue":"cs.LG","work_id":"94cb6c9f-d0ab-44e9-b66e-7397f3a9f117","year":2025},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2509.03403","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:472d8fd18bf75ca61319c4be7b39fdffa68138345cc9eebe3d1351fba90ee9b2","observation_id":"a2104129-86f3-4840-ba2f-df6f312ef891","resolution":{"observed_at":"2026-06-30T07:34:21.178202Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.09459","last_updated":"2026-04-13T12:08:22Z","snapshot_observed_at":"2026-07-06T22:58:21.624968Z","submitted_at":"2026-04-10T16:17:44Z","title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","version":2},"cited_work":{"arxiv_id":"2604.09459","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.09459","snapshot_observed_at":"2026-07-09T11:16:11.379473Z","title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","venue":"cs.CL","work_id":"b28c3265-685a-4be5-b6e3-cfcd46733d99","year":2026},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2604.09459","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:d191bebdfe475fbba58b5c25d7af8b4f2cb4bcf3aff5cfe6306b7edb1ccfbca7","observation_id":"61182028-d306-4652-8739-f1b2efda3b50","resolution":{"observed_at":"2026-06-30T07:34:21.175163Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.10535","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T07:34:21.199793Z","title":"Tackling Length Inflation Without Trade-offs: Group Relative Reward Rescaling","venue":null,"work_id":"8d6247e7-2040-4d30-a1d0-6cd08e0ceb97","year":2026},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:bdd720b60ebdf9e14a68d0c01c22ba3b9637cd19a267e8dded8000e49a6b86bd","observation_id":"b5141abb-d49c-4149-a2f3-10b441144e9d","resolution":{"observed_at":"2026-06-30T07:34:21.201958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.19835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T03:29:30.724740Z","title":"Fipo: Eliciting deep reasoning with future-kl influenced policy optimization","venue":null,"work_id":"5f612bbe-1204-4f05-b54d-9a02bc9d66ea","year":2026},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:5ed0d53787045c49559e32f55fc05e7941fed973e462f252fed2b92c90ffc64c","observation_id":"9121599e-67e2-4b46-b717-87f431398145","resolution":{"observed_at":"2026-06-30T07:34:21.183878Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.09331","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T07:34:21.206266Z","title":"Junxi Yin, Haisen Luo, Zhenyu Li, Yihua Liu, Dan Liu, Zequn Li, and Xiaohang Xu","venue":null,"work_id":"c6fee1be-74b0-4a22-83c2-d12ff53f7f54","year":null},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:d4f567f27e8367f6a61f5d6031447f30a5bf46b66ee336639f8d106239e09730","observation_id":"2089c9ec-877d-485d-a7de-9a276d833b30","resolution":{"observed_at":"2026-06-30T07:34:21.208021Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.08899","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T07:34:21.196699Z","title":"Pinpointing crucial steps: Attribution-based credit assignment for verifiable reinforcement learning.arXiv preprint arXiv:2510.08899","venue":null,"work_id":"d9c88624-f452-4e63-9a40-4851654d9330","year":null},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:6729151d6461c9bf1e4a188441a5c870b18978c163295c155fabc8503be9e07b","observation_id":"d0a1ea47-ae02-4361-a8f3-bb93f2e628c7","resolution":{"observed_at":"2026-06-30T07:34:21.198548Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.14987","last_updated":"2022-03-28T18:00:51Z","snapshot_observed_at":"2026-07-06T12:53:50.496350Z","submitted_at":"2022-03-28T18:00:51Z","title":"Multilingual Knowledge Graph Completion with Self-Supervised Adaptive Graph Alignment","version":1},"cited_work":{"arxiv_id":"2203.14987","doi":"10.48550/arxiv","metadata_source":"doi_reference","pith_arxiv_id":"2203.14987","snapshot_observed_at":"2026-07-09T21:46:34.506658Z","title":"Dickerson","venue":"cs.AI","work_id":"5c2060c6-427c-4321-be22-49ccae439d80","year":2025},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2203.14987","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:6d44f5904f9b09aa0e7ccae14b3483bd3ba03749ed1194b1a409bc6f49dd239b","observation_id":"c43312f5-f21f-47e1-9ba6-50aae585c62f","resolution":{"observed_at":"2026-06-30T07:34:20.679855Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.04474","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:19:43.845608Z","title":"Drpo: Efficient reasoning via decoupled reward policy optimization","venue":null,"work_id":"9841afa4-9ee4-487f-9af4-d8f4bcf6b782","year":2025},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:da9d39e07b75313220b60eecb0d6da1aa6e47369dd27631fbdc1e05af8a93544","observation_id":"d9a15bad-183c-43fa-90d7-579b34f4bb28","resolution":{"observed_at":"2026-06-30T07:34:21.205151Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.01925","last_updated":"2026-08-03T11:37:46Z","snapshot_observed_at":"2026-08-06T23:28:17.380011Z","submitted_at":"2025-10-02T11:42:17Z","title":"Enhancing Large Language Model Reasoning with Reward Models: An Analytical Survey","version":3},"cited_work":{"arxiv_id":"2510.01925","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2510.01925","snapshot_observed_at":"2026-08-04T02:39:30.770950Z","title":"Enhancing large language model reasoning with reward models: An analytical survey","venue":null,"work_id":"8c8d8d11-c25e-43b3-aa44-20b8338f989e","year":2025},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2510.01925","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:864b9aa68be2946fce5015077146b5fe81d6a1b0c3499447b7c83addbbc075f0","observation_id":"164d8e26-45ef-4cf3-8df2-4949f61a554c","resolution":{"observed_at":"2026-08-04T02:39:30.770950Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12125","last_updated":"2026-02-26T13:26:22Z","snapshot_observed_at":"2026-07-31T18:08:52.441829Z","submitted_at":"2026-02-12T16:14:29Z","title":"Learning beyond Teacher: Generalized On-Policy Distillation with Reward Extrapolation","version":2},"cited_work":{"arxiv_id":"2602.12125","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.12125","snapshot_observed_at":"2026-07-09T00:35:48.740330Z","title":"Learning beyond Teacher: Generalized On-Policy Distillation with Reward Extrapolation","venue":"cs.LG","work_id":"bb968107-1f43-4bf4-aa52-cc58000a6e89","year":2026},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2602.12125","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:027d7f4852dce661f2f03255b6cf68982554bf98e2f07d713299d45c35c02f1b","observation_id":"21ccff55-542f-47bb-8b8e-6372587e1828","resolution":{"observed_at":"2026-06-30T07:34:21.186840Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":"2409.12122","doi":"10.18653/v1/2025.emnlp-main.712","metadata_source":"pith","pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","venue":"cs.CL","work_id":"a097c5d4-6d32-46ee-9826-57d532bbfc9c","year":2024},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:a782a7d6a1009bdfde2243e4ac16c2631cd82c349b0457bebc9b1df1a438ebaa","observation_id":"5c210f49-f540-4104-a341-0d2b6abc61a3","resolution":{"observed_at":"2026-06-30T07:34:21.192165Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T07:33:55.284758Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:432b7796931e26828b2043b882180540cb727917fcecdd48977aa92a6c55bdfb","observation_id":"6311e93f-6bf9-4425-b939-c5455ad5fd4c","resolution":{"observed_at":"2026-06-30T07:33:55.284758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2020.coling-main.580","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T19:08:49.111773Z","title":"Constructing A Multi-hop QA Dataset for Comprehensive Evaluation of Reasoning Steps","venue":"Proceedings of the 28th International Conference on Computational Linguistics","work_id":"fd3a2c44-aeea-48a7-b162-d3f0c6d43f35","year":2020},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:c41f3cd5abf49e3ed60447d0a1fd2ea4254a40cff3482f8ecdaa7dff14f38653","observation_id":"0f91f9df-6b75-48ce-a74c-fe82f04854bc","resolution":{"observed_at":"2026-06-30T07:34:20.683703Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T05:50:02.713939+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T05:50:02.713939+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1162/tacl_a_00475","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T00:19:12.580196Z","title":"♫ M u S i Q ue: Multihop Questions via Single-hop Question Composition","venue":"Transactions of the Association for Computational Linguistics","work_id":"dd4f6eb0-477f-42c4-8d8a-8c8637815f98","year":2022},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:ab5887611b83e37bc3af893a99b8b5fef0c72eac8f49553878cadf9db64f5ceb","observation_id":"af4dd9d7-5f6a-49ad-963d-d50f7f2a66b5","resolution":{"observed_at":"2026-06-30T07:34:20.689750Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T05:50:02.462802+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T05:50:02.462802+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.15115","doi":"10.1145/3581783.3612503","metadata_source":"pith","pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5 Technical Report","venue":"cs.CL","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","year":2024},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:6f4e4ef6320d0779026f765f425bb258ec937521bc92b6f87af150f3fb12383f","observation_id":"98ac186d-3007-4402-a640-ff7c727d445f","resolution":{"observed_at":"2026-06-30T07:34:21.189518Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T07:33:55.284758Z","title":"Setting k<1.0 protects necessary reasoning verbosity and raises the exploratory ceiling; k=0.7 achieves the best average pass@1 and is used as the default throughout this paper","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-30T07:33:55.284758Z"},"links":{"citing_paper":"/paper/2606.29296"},"observation_digest":"sha256:e4e93c68edf980886b4f1995e59402e521fd1ece6449afdd76a638e4ba057106","observation_id":"35ecaa5c-aadf-4b27-a957-aff9c423f466","resolution":{"observed_at":"2026-06-30T07:33:55.284758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.29296","last_updated":"2026-06-28T09:36:43Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-07T00:03:16.968118Z","submitted_at":"2026-06-28T09:36:43Z","title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners"},"reference_resolution":{"displayed":18,"state_counts":{"malformed_identifier":1,"metadata_mismatch":9,"parse_uncertain":0,"unresolved":2,"verified_exact":6,"verified_fuzzy":0},"total_outbound_references":18},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 18 of 18 outbound references and 0 inbound Pith citation observations for arXiv:2606.29296."}