{"as_of":"2026-08-08T08:28:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dec6434774e6c61edb629dba091e2abfc7946a873f73638dc28571989814de06","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:06:42.841133Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T19:57:32.948217Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-01T20:56:13.479928Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.03570","snapshot_observed_at":"2026-08-02T19:57:32.948217Z","title":"Freeprm: Training process reward mod- els without ground truth process labels.arXiv preprint arXiv:2506.03570, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.06652","last_updated":"2026-06-11T03:22:36Z","snapshot_observed_at":"2026-08-07T18:35:30.659839Z","submitted_at":"2026-02-28T04:33:11Z","title":"PaLMR: Towards Faithful Visual Reasoning via Multimodal Process Alignment","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T19:57:32.948217Z"},"links":{"cited_paper":"/paper/2506.03570","citing_paper":"/paper/2603.06652"},"observation_digest":"sha256:1c36fda67c1f12ac90956531aeb62639877b00eeb6907f933c7b4f9cb20604dd","observation_id":"355ce1e2-f6ca-4d58-983a-4b788d6b6edb","resolution":{"observed_at":"2026-08-02T19:57:32.948217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"cited_work":{"arxiv_id":"2506.03570","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.03570","snapshot_observed_at":"2026-07-01T20:56:13.479928Z","title":"Freeprm: Training process reward models without ground truth process labels","venue":null,"work_id":"5ed01953-7776-4339-b9ea-1c782a9b6933","year":2025},"citing_paper":{"arxiv_id":"2605.10158","last_updated":"2026-05-11T08:05:27Z","snapshot_observed_at":"2026-08-06T12:54:50.132575Z","submitted_at":"2026-05-11T08:05:27Z","title":"Unsupervised Process Reward Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T03:19:04.275069Z"},"links":{"cited_paper":"/paper/2506.03570","citing_paper":"/paper/2605.10158"},"observation_digest":"sha256:a328d22009fda93f98821e895b95e5c4a87bf4929bcc1477fd966479b4540a71","observation_id":"d1a55f1a-dc7c-49ae-9cea-5bfb614c2268","resolution":{"observed_at":"2026-05-12T03:21:18.892131Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"cited_work":{"arxiv_id":"2506.03570","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.03570","snapshot_observed_at":"2026-07-01T20:56:13.479928Z","title":"Freeprm: Training process reward models without ground truth process labels","venue":null,"work_id":"5ed01953-7776-4339-b9ea-1c782a9b6933","year":2025},"citing_paper":{"arxiv_id":"2605.15529","last_updated":"2026-05-15T01:57:11Z","snapshot_observed_at":"2026-08-04T04:14:15.447935Z","submitted_at":"2026-05-15T01:57:11Z","title":"Process Rewards with Learned Reliability","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-19T14:51:13.966538Z"},"links":{"cited_paper":"/paper/2506.03570","citing_paper":"/paper/2605.15529"},"observation_digest":"sha256:0714b10a093108df68d54c6d63a30b7bd5f325ab49b42931375f6c8f762cf4fc","observation_id":"2dc7b6a4-248e-44e1-b692-e4fa63a8c41b","resolution":{"observed_at":"2026-05-19T14:53:06.933322Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"cited_work":{"arxiv_id":"2506.03570","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.03570","snapshot_observed_at":"2026-07-01T20:56:13.479928Z","title":"Freeprm: Training process reward models without ground truth process labels","venue":null,"work_id":"5ed01953-7776-4339-b9ea-1c782a9b6933","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2506.03570","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:246e4bd8661ffbbb2939b2958d3af661d800a84f1dc35f2c560f0db4b3a18680","observation_id":"32756f27-4328-4fa0-ac5b-0dd79cc3f4d8","resolution":{"observed_at":"2026-07-01T20:56:13.481368Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.03570/citation-record","integrity":"/paper/2506.03570/integrity","json":"/paper/2506.03570/citation-record.json","paper":"/paper/2506.03570"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:44.601936Z","title":"Alphamath almost zero: Process super- vision without process","venue":null,"work_id":"eb4ded1d-9dc7-4c11-b10f-3182751148fc","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:40.848523Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:a1aa4d6b71d60d60dd682efac1ed643d6ad3280da1be13b07aee000e9ca812ca","observation_id":"b692a8df-3cdc-4c89-be02-b60657262463","resolution":{"observed_at":"2026-08-07T11:06:44.644217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T11:06:40.892903Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:40.892903Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:bd95580e3ac74188809d8d93ad895986c34258be361b7d94f7d246c9f9d91ac6","observation_id":"1f6013f9-299f-4d41-8515-9aba3e27ce84","resolution":{"observed_at":"2026-08-07T11:06:40.892903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01456","last_updated":"2025-09-26T09:25:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-03T15:43:48Z","title":"Process Reinforcement through Implicit Rewards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01456","snapshot_observed_at":"2026-08-07T11:06:40.984203Z","title":"Process reinforcement through implicit rewards.arXiv preprint arXiv:2502.01456, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:40.984203Z"},"links":{"cited_paper":"/paper/2502.01456","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:03bdf4885402b0575a511b8d94f263e07bfd1a337b711ba5310102cf130aa14b","observation_id":"409cec1e-f603-4b34-9b8c-778ef84edb9a","resolution":{"observed_at":"2026-08-07T11:06:40.984203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T11:06:41.062201Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.062201Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:de5fdb4cc091c26a2f2912d29a762fe8189cfa5fe4bb6b7eda94504fceb0f6ac","observation_id":"ce41dbb2-d935-40d8-917c-c8f0874ef52c","resolution":{"observed_at":"2026-08-07T11:06:41.062201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14024","last_updated":"2024-10-18T06:59:24Z","snapshot_observed_at":"2026-08-06T13:33:28.979461Z","submitted_at":"2024-06-20T06:42:27Z","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14024","snapshot_observed_at":"2026-08-07T11:06:41.126923Z","title":"LLM critics help catch bugs in mathematics: Towards a better mathematical verifier with natural language feedback","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.126923Z"},"links":{"cited_paper":"/paper/2406.14024","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:064d5443c34312ac4cb8acbc3171ddb1316889f27f420679877a835c5bd22aa4","observation_id":"60386761-00f8-4d14-9105-f477aa201bba","resolution":{"observed_at":"2026-08-07T11:06:41.126923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:44.499732Z","title":"Measuring mathematical problem solving with the MATH dataset","venue":null,"work_id":"572d0c5c-bf73-4992-8041-09a1def99dea","year":2021},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.206238Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:d446c4a9dd3873e1308345e52e278cb4822aa9e975a838b05c18f31ec9b390d0","observation_id":"ae69aec8-0373-439b-ab6c-ce25584e52db","resolution":{"observed_at":"2026-08-07T11:06:44.562843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-07T11:06:41.250937Z","title":"Qwen2.5-coder technical report.arXiv preprint arXiv:2409.12186, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.250937Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:91ad96f8ebf4ff3d46f792022d647d9269ef25fd54ac37c2f5cbcb905a2129f8","observation_id":"07e9f2ba-67cb-4956-8cec-c8d6ef4c59e1","resolution":{"observed_at":"2026-08-07T11:06:41.250937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-07T11:06:41.343356Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.343356Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:761b9f8f9c176a9f898e624370f9e88043e23908eb6f79f82f92710abed8d13c","observation_id":"5d6ddf36-bc9b-479f-b39c-69e0daab2ff2","resolution":{"observed_at":"2026-08-07T11:06:41.343356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11287","last_updated":"2025-02-11T05:41:41Z","snapshot_observed_at":"2026-08-06T19:03:24.746489Z","submitted_at":"2024-10-15T05:10:34Z","title":"Process Reward Model with Q-Value Rankings","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11287","snapshot_observed_at":"2026-08-07T11:06:41.397499Z","title":"Process reward model with q-value rankings.arXiv preprint arXiv:2410.11287, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.397499Z"},"links":{"cited_paper":"/paper/2410.11287","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:7b44faefcd5e7393247775594cf4706fe59a98911eecd852766aef002de9531e","observation_id":"d879184b-5a0e-4540-b615-95c2dd7eec5d","resolution":{"observed_at":"2026-08-07T11:06:41.397499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:44.385548Z","title":"Let’s verify step by step","venue":null,"work_id":"cf5b2924-ea19-4f02-ae27-0909612505c6","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.502360Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:77c2b133f86e010a54c21c7a2462afa2f5c35e175a490e42df33b3fb61d131fe","observation_id":"41423e53-92bb-439f-b9d5-05f4325a7b4b","resolution":{"observed_at":"2026-08-07T11:06:44.434667Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:44.272105Z","title":"Autopsv: Automated process-supervised verifier","venue":null,"work_id":"799bf86a-257b-452e-9f1f-112a0032902f","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.560493Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:2f4cfc97f305e44ea63a130c24fca1e768619628bf14a47daed5046f83fbd59e","observation_id":"768ef0a6-eb07-4df4-ba05-4b74f7a63012","resolution":{"observed_at":"2026-08-07T11:06:44.321785Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06592","last_updated":"2024-12-11T22:59:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-05T19:25:40Z","title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06592","snapshot_observed_at":"2026-08-07T11:06:41.654194Z","title":"Improve mathematical reasoning in language models by automated process supervision.arXiv preprint arXiv:2406.06592, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.654194Z"},"links":{"cited_paper":"/paper/2406.06592","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:b031deb5d3342b02cd3bea01a8e7fd1736d2fe3ae01fd3ce8f7f114729df3a01","observation_id":"71f147f5-9355-44db-9ffd-cf487da4571e","resolution":{"observed_at":"2026-08-07T11:06:41.654194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-07T11:06:41.718662Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.718662Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:3f783264d747ad5c2d157154edc8aa6a5c964469d5432e45bb82b7fcfdfbf3ef","observation_id":"2107b35b-0ad3-45ea-b7eb-01f31268c4ff","resolution":{"observed_at":"2026-08-07T11:06:41.718662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:44.142683Z","title":"Skywork-o1 open series","venue":null,"work_id":"e3c77666-6669-4846-9d48-6f517e784363","year":null},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.762843Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:55851b01f0fc155df261a2bc84610bbacb754031f83c0d0d360b778d26ddecf6","observation_id":"c138dd29-192a-467a-aada-eb6e5c7cabaa","resolution":{"observed_at":"2026-08-07T11:06:44.213012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-07T11:06:41.911560Z","title":"Scaling LLM test-time compute op- timally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.911560Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:a91bf2309f999bca8d09adc9cdd430bddf04df93b771b44acdf0e2cf2de4db76","observation_id":"b94bc281-9988-4294-afbb-d713b7ea1435","resolution":{"observed_at":"2026-08-07T11:06:41.911560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14283","last_updated":"2024-07-22T10:01:49Z","snapshot_observed_at":"2026-07-06T18:34:15.291938Z","submitted_at":"2024-06-20T13:08:09Z","title":"Q*: Improving Multi-step Reasoning for LLMs with Deliberative Planning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14283","snapshot_observed_at":"2026-08-07T11:06:42.008548Z","title":"Q*: Improving multi-step reasoning for llms with deliberative planning.arXiv preprint arXiv:2406.14283, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.008548Z"},"links":{"cited_paper":"/paper/2406.14283","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:63340e5e25cdeaf967ef09ed598ce4fa62b3c4cdd5d57ca14a222c2108733728","observation_id":"e88e6dc7-a111-4a8e-8439-e4c2192d5cd3","resolution":{"observed_at":"2026-08-07T11:06:42.008548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.899811Z","title":"Math-shepherd: Verify and reinforce llms step-by-step without human annotations","venue":null,"work_id":"91608c92-c99d-4fc9-8f1d-84e7a147c95f","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.057605Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:6d463aa7a2ae8f9353bf9a87f23475ca88d413a8636254f3bb60ebfc363799fa","observation_id":"67cac0ea-15ea-4643-b7c7-f04d64faecf2","resolution":{"observed_at":"2026-08-07T11:06:43.963514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.800611Z","title":"Multi-step problem solving through a verifier: An empirical analysis on model-induced process supervision","venue":null,"work_id":"3b8be63e-4e3f-4432-bfeb-65114ab91f0d","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.140691Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:3d3be234c578e5bd09baf6c2564d69e68dad680745be025d9b5f3db906db805b","observation_id":"9a29aa4f-7abc-4e3f-89ef-5b847debc3a2","resolution":{"observed_at":"2026-08-07T11:06:43.838113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.689014Z","title":"Training large language models for reasoning through reverse curriculum reinforcement learning","venue":null,"work_id":"599859f4-3326-41d6-9316-ba7e8ccce76f","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.205649Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:903e296bf29a426d2977aa23b7868d11624d9adb4e3190e32cd013a0aeeb3875","observation_id":"67831e97-c489-455b-9ff2-72921486e6ab","resolution":{"observed_at":"2026-08-07T11:06:43.758576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.592327Z","title":"Evaluating mathematical reasoning beyond accuracy","venue":null,"work_id":"f1a4f554-8236-4aae-bf93-db24c3a14c51","year":2025},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.272829Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:e115894f6b90703536a285e9f6b37cbc2ee8ad20b8f7fe3d622077ee5e7cf1c2","observation_id":"e37d8d9a-6eea-411f-a8c9-0c9fc9a8503f","resolution":{"observed_at":"2026-08-07T11:06:43.633494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.435877Z","title":"An implementation of generative prm","venue":null,"work_id":"b8869f0a-9fc2-4e49-bac0-36e582375087","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.339628Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:2317f023f55fa9f3b74de2c1edc59d601fe1003719f27a753df420ebb7a44f5b","observation_id":"f102bf8e-4392-41a8-91f9-5159c003ec7f","resolution":{"observed_at":"2026-08-07T11:06:43.509443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T11:06:42.395649Z","title":"Qwen2.5 technical report.arXiv preprint arXiv:2412.15115, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.395649Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:eba08e82adb31670dac13030a0cca8bfab414acd3984bbbed4bc081240ce92c8","observation_id":"cbc7beb3-2e64-4474-9d5b-61baa3ed4d88","resolution":{"observed_at":"2026-08-07T11:06:42.395649Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12122","last_updated":"2024-09-18T16:45:37Z","snapshot_observed_at":"2026-07-06T19:17:41.512834Z","submitted_at":"2024-09-18T16:45:37Z","title":"Qwen2.5-Math Technical Report: Toward Mathematical Expert Model via Self-Improvement","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12122","snapshot_observed_at":"2026-08-07T11:06:42.449307Z","title":"Qwen2.5-math technical report: Toward mathematical expert model via self-improvement.arXiv preprint arXiv:2409.12122, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.449307Z"},"links":{"cited_paper":"/paper/2409.12122","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:288909e34ffbb49188e716f673f15083cba1a99f44b3fd2eb1cdb814c6e1bcab","observation_id":"a586e8ac-6143-45c3-a3dc-3f523d1d4f41","resolution":{"observed_at":"2026-08-07T11:06:42.449307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.261820Z","title":"Ovm, outcome-supervised value models for planning in mathematical reasoning","venue":null,"work_id":"ad4e0181-0d01-4a17-83ad-4b7a00f7c4d5","year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.539085Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:a7315b9ee852e59ae8101baf1c4c41b05eb3a84dc2665bf08f4694cd4fb02c98","observation_id":"64cba8e6-cf3c-4a15-a310-e833bd8a7636","resolution":{"observed_at":"2026-08-07T11:06:43.348491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01981","last_updated":"2024-12-02T21:20:02Z","snapshot_observed_at":"2026-08-06T03:48:05.506841Z","submitted_at":"2024-12-02T21:20:02Z","title":"Free Process Rewards without Process Labels","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01981","snapshot_observed_at":"2026-08-07T11:06:42.575866Z","title":"Free process rewards without process labels.arXiv preprint arXiv:2412.01981, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.575866Z"},"links":{"cited_paper":"/paper/2412.01981","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:cc70ffecf555c60dadd2b37eeed9c3a60492d3e41d72655eebf0a9cc7c43defe","observation_id":"aef94d07-53cf-45ff-8dfe-8b2e80688aae","resolution":{"observed_at":"2026-08-07T11:06:42.575866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15240","last_updated":"2025-02-22T10:21:46Z","snapshot_observed_at":"2026-07-06T19:06:41.697671Z","submitted_at":"2024-08-27T17:57:45Z","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15240","snapshot_observed_at":"2026-08-07T11:06:42.642734Z","title":"Generative verifiers: Reward modeling as next-token prediction.arXiv preprint arXiv:2408.15240, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.642734Z"},"links":{"cited_paper":"/paper/2408.15240","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:53adc2637246a20a9a60c38d2ee733b634477a66bdb185417c2ce4b4ffd1b93b","observation_id":"944cfdcc-93f7-47b8-82ea-1109dd1f95d8","resolution":{"observed_at":"2026-08-07T11:06:42.642734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07301","last_updated":"2025-06-05T16:34:24Z","snapshot_observed_at":"2026-08-03T11:11:25.359494Z","submitted_at":"2025-01-13T13:10:16Z","title":"The Lessons of Developing Process Reward Models in Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07301","snapshot_observed_at":"2026-08-07T11:06:42.722080Z","title":"The lessons of developing process reward models in mathematical reasoning.arXiv preprint arXiv:2501.07301, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.722080Z"},"links":{"cited_paper":"/paper/2501.07301","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:4e985748f5e6333ceb8272eec3dd3f8a5a8e5e0e68b924d796a9ad5c8fa04730","observation_id":"814b695d-53dc-4b7b-b30c-487d35b11c74","resolution":{"observed_at":"2026-08-07T11:06:42.722080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06559","last_updated":"2025-05-26T14:03:32Z","snapshot_observed_at":"2026-08-05T20:30:49.812919Z","submitted_at":"2024-12-09T15:11:40Z","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06559","snapshot_observed_at":"2026-08-07T11:06:42.769583Z","title":"Limitations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.769583Z"},"links":{"cited_paper":"/paper/2412.06559","citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:90371f276edebd5095b215d44a6c4eb23cf2acd15fe83bb99d015a8893237977","observation_id":"63478cf3-b4f8-4893-85c5-e0d097331ed0","resolution":{"observed_at":"2026-08-07T11:06:42.769583Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:43.098986Z","title":"Guidelines: • The answer NA means that the paper does not involve crowdsourcing nor research with human subjects","venue":null,"work_id":"e40cc30b-d09d-4e1b-8eae-7a072a935c2d","year":2025},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:42.841133Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:5ddae965ba12bcecd32dac13c4d8579b818ef87d595f1d304ce3d1ca7c7c456d","observation_id":"62e255ff-a0d4-48b2-8583-4ac3d54511ee","resolution":{"observed_at":"2026-08-07T11:06:43.161376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:06:44.025401Z","title":null,"venue":null,"work_id":"ba0bc149-0cea-4876-959a-e3c9f701621e","year":null},"citing_paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T11:06:41.854876Z"},"links":{"citing_paper":"/paper/2506.03570"},"observation_digest":"sha256:ee08345d2282700fa82f04dbebffeb324c97a9a13196569e90d2efdf592f42e5","observation_id":"8e862179-7b9f-476a-ab48-33c92b429597","resolution":{"observed_at":"2026-08-07T11:06:44.086411Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.03570","last_updated":"2025-06-04T04:33:53Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T10:57:29.750745Z","submitted_at":"2025-06-04T04:33:53Z","title":"FreePRM: Training Process Reward Models Without Ground Truth Process Labels"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":12},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 4 inbound Pith citation observations for arXiv:2506.03570."}