{"as_of":"2026-08-06T13:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c475c361d44f9c83ea93a79646ed0e383577ea72e76eab47da33f4c95613701d","coverage":[{"denominator":29,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T23:24:43.556606Z","state":"measured"},{"denominator":32,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":32,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-30T11:06:21.527926Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-06-30T11:54:38.713100Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":"2507.15698","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.15698","snapshot_observed_at":"2026-06-30T11:54:38.713100Z","title":"Cold: Counterfactually-guided length debiasing for process reward models","venue":"cs.CL","work_id":"21fecb2e-5e3a-4629-8b21-dc8b5ff7012e","year":2025},"citing_paper":{"arxiv_id":"2510.14703","last_updated":"2026-04-28T18:17:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-16T14:06:03Z","title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-18T06:30:39.858246Z"},"links":{"cited_paper":"/paper/2507.15698","citing_paper":"/paper/2510.14703"},"observation_digest":"sha256:386ac76dd7bec6f0ca57a324930dca215fd2f9407dc92c24a12282fe6c8aea24","observation_id":"a3e94bf0-9133-432f-add4-393ededb352e","resolution":{"observed_at":"2026-05-20T02:04:43.676793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":"2507.15698","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.15698","snapshot_observed_at":"2026-06-30T11:54:38.713100Z","title":"Cold: Counterfactually-guided length debiasing for process reward models","venue":"cs.CL","work_id":"21fecb2e-5e3a-4629-8b21-dc8b5ff7012e","year":2025},"citing_paper":{"arxiv_id":"2604.13602","last_updated":"2026-04-15T08:11:34Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T08:11:34Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T13:58:53.430492Z"},"links":{"cited_paper":"/paper/2507.15698","citing_paper":"/paper/2604.13602"},"observation_digest":"sha256:fbbee2cd801b577b23364114566f427a67d49fa43fa3747fb87e25737de5b2c0","observation_id":"533848bc-7e3c-49a6-9437-236c1e7ff9b4","resolution":{"observed_at":"2026-05-20T02:04:43.676793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":"2507.15698","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.15698","snapshot_observed_at":"2026-06-30T11:54:38.713100Z","title":"Cold: Counterfactually-guided length debiasing for process reward models","venue":"cs.CL","work_id":"21fecb2e-5e3a-4629-8b21-dc8b5ff7012e","year":2025},"citing_paper":{"arxiv_id":"2605.25143","last_updated":"2026-05-31T05:24:41Z","snapshot_observed_at":"2026-07-06T23:35:06.733347Z","submitted_at":"2026-05-24T15:48:57Z","title":"Beyond the Frontier: Stochastic Backtracking for Efficient Test-Time Scaling","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-30T11:06:21.527926Z"},"links":{"cited_paper":"/paper/2507.15698","citing_paper":"/paper/2605.25143"},"observation_digest":"sha256:ebcc1e33cec4c14c1c3531c285def7ac1fe11247167e9dc8d954f5ab19603c8d","observation_id":"95d3a8f1-771d-47c0-ae0c-9673fba8a254","resolution":{"observed_at":"2026-06-30T11:54:38.714362Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.15698/citation-record","integrity":"/paper/2507.15698/integrity","json":"/paper/2507.15698/citation-record.json","paper":"/paper/2507.15698"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:7b36f2b8fa026273d90683974e47d8f1fa763bd0bf1384db41c75f49dda412ee","observation_id":"5b2b89f6-d0dd-4ee0-b31e-116eee59ea4d","resolution":{"observed_at":"2026-05-21T23:25:45.368154Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07319","last_updated":"2024-02-11T22:40:12Z","snapshot_observed_at":"2026-07-06T17:28:39.754296Z","submitted_at":"2024-02-11T22:40:12Z","title":"ODIN: Disentangled Reward Mitigates Hacking in RLHF","version":1},"cited_work":{"arxiv_id":"2402.07319","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07319","snapshot_observed_at":"2026-07-03T17:58:46.669571Z","title":"Odin: Disentangled reward mitigates hacking in rlhf","venue":null,"work_id":"a8dcc440-de24-436b-835e-e83044fcff08","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2402.07319","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:a9cf072abed5c1f7ac23e745db9c461574793443ae0006caa1152465809feb95","observation_id":"163fec65-6cc6-462f-b789-08baac871e75","resolution":{"observed_at":"2026-05-21T23:25:45.371973Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:7a83bea741fdf5271d653a692ac607c474664290a3324ad319599d6b1e09f834","observation_id":"6de869e4-ab8d-4720-967e-6ff994714573","resolution":{"observed_at":"2026-05-21T23:25:45.365176Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09244","last_updated":"2024-08-16T23:59:29Z","snapshot_observed_at":"2026-07-06T17:02:05.607732Z","submitted_at":"2023-12-14T18:59:04Z","title":"Helping or Herding? Reward Model Ensembles Mitigate but do not Eliminate Reward Hacking","version":3},"cited_work":{"arxiv_id":"2312.09244","doi":"10.48550/arxiv.2312.09244","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.09244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"J., Fisch, A., Heller, K., Pfohl, S., Ramachandran, D., Shaw, P., and Berant, J","venue":"arXiv (Cornell University)","work_id":"0d3df61b-eeb7-4a01-98a5-de379e1ac583","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2312.09244","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:b34201e8716030c4adfe3375d7d1214ca1fa7962c9ca3f56d47112683edef981","observation_id":"9d88245e-cb38-489a-962f-89e09abc6976","resolution":{"observed_at":"2026-05-21T23:25:45.324401Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-22T21:23:20.318631+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T21:23:20.318631+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14024","last_updated":"2024-10-18T06:59:24Z","snapshot_observed_at":"2026-08-06T13:33:28.979461Z","submitted_at":"2024-06-20T06:42:27Z","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","version":4},"cited_work":{"arxiv_id":"2406.14024","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.14024","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv:2406.14024","venue":null,"work_id":"b8a7626e-ea9b-43ef-b087-69fd533b7413","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2406.14024","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:2596f1805175c1353a12d723224acd7984eb6cb3f3acd303d1b36344d4538ee9","observation_id":"f0457bb7-171b-43e5-998b-d3f28515fca2","resolution":{"observed_at":"2026-05-21T23:25:45.332497Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":"2103.03874","doi":"10.48550/arxiv.2103.03874","metadata_source":"pith","pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","venue":"cs.LG","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","year":2021},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:3aa378907f69db9c891c7a3165c58f12ca97ad8641462c1ce11bc2b2a1fa3997","observation_id":"832db875-eaa1-4760-875a-c1ea9ec09a22","resolution":{"observed_at":"2026-05-21T23:25:45.316700Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.649941+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2409.17407","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T19:40:07.140762Z","title":"Chen et al","venue":null,"work_id":"eff993c8-779e-4a80-8118-2bca35735296","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:2b6fb09aea8e7fc013f6d5517cb423f765eff517a306eabcabf4a28bc2ac5446","observation_id":"6dfc41c7-39f8-4962-a5ff-d7e3e45a9360","resolution":{"observed_at":"2026-05-21T23:25:45.352971Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19255","last_updated":"2024-07-02T03:46:03Z","snapshot_observed_at":"2026-08-06T12:10:50.595602Z","submitted_at":"2024-02-29T15:26:14Z","title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","version":2},"cited_work":{"arxiv_id":"2402.19255","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.19255","snapshot_observed_at":"2026-07-03T22:39:01.102254Z","title":"Li, W.; and Li, Y","venue":null,"work_id":"e9445936-9002-4497-ad10-984bf294269f","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2402.19255","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:6b5cf52feb6e79cdf46bc19f05fd4b4f2a86cf894f29496241e14166f29d8a3c","observation_id":"b1e3dcf4-64f9-4e84-bc91-83c1113b8d8d","resolution":{"observed_at":"2026-05-21T23:25:45.340074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11287","last_updated":"2025-02-11T05:41:41Z","snapshot_observed_at":"2026-07-06T19:33:39.134036Z","submitted_at":"2024-10-15T05:10:34Z","title":"Process Reward Model with Q-Value Rankings","version":2},"cited_work":{"arxiv_id":"2410.11287","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.11287","snapshot_observed_at":"2026-07-04T13:59:51.866161Z","title":"arXiv preprint arXiv:2410.11287 , year=","venue":null,"work_id":"7a6628a4-ed2b-4203-8e89-78bc486b018e","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2410.11287","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:50431e877e2ebde2e60f35809b283698bd2b65bebccfdb41b7d4afed3bc68f44","observation_id":"1dff3c38-1b04-47eb-a071-2298d56aea64","resolution":{"observed_at":"2026-05-21T23:25:45.336123Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":"2305.20050","doi":"10.1007/bf00262952","metadata_source":"pith","pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Let's Verify Step by Step","venue":"cs.LG","work_id":"6d05b790-04c5-4fd2-91b2-ba1dfdd5770f","year":2023},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:d9fb0a1986875d7c3931aeee618c559219af8798efae113e8d47a02f75fabd9d","observation_id":"024e02c0-bee5-4045-8315-4b0e7c3c85a7","resolution":{"observed_at":"2026-05-21T23:25:45.271500Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.19437","doi":"10.1016/j.neucom.2023.127063.url:https://www.sciencedirect","metadata_source":"pith","pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeek-V3 Technical Report","venue":"cs.CL","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:cae6dd11e2063d1b16965f41aa6f53edf11828b2e3e41cd7fc4fb622cbd2a760","observation_id":"ed1b5cf5-91ac-4b26-9453-d01605e7cf36","resolution":{"observed_at":"2026-05-21T23:25:45.282531Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06592","last_updated":"2024-12-11T22:59:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-05T19:25:40Z","title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","version":2},"cited_work":{"arxiv_id":"2406.06592","doi":"10.48550/arxiv.2406.06592","metadata_source":"pith","pith_arxiv_id":"2406.06592","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","venue":"cs.CL","work_id":"9906447b-5bb3-4367-91f9-b9d556c9ee7f","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2406.06592","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:4dd84c5899d4d4933a207fac63ad6942593c835445199f1eea3f75f3de68d096","observation_id":"0d27e7ee-a36b-4ce5-9891-32da91502027","resolution":{"observed_at":"2026-05-21T23:25:45.279073Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00215","last_updated":"2024-06-28T19:53:17Z","snapshot_observed_at":"2026-07-06T18:38:46.314431Z","submitted_at":"2024-06-28T19:53:17Z","title":"LLM Critics Help Catch LLM Bugs","version":1},"cited_work":{"arxiv_id":"2407.00215","doi":"10.48550/arxiv.2407.00215","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.00215","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLM Critics Help Catch LLM Bugs","venue":"arXiv (Cornell University)","work_id":"e730c1aa-4c9b-48f3-923b-47ddc0a4e6d8","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2407.00215","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:8bf48b274e231212f4425774c9a5935a477211f7e14bd564450ba353d8962bfb","observation_id":"9adddc11-31b6-4e55-a087-ba9a64a6eb06","resolution":{"observed_at":"2026-05-21T23:25:45.268233Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:c4d7a44971b02d92f98653558fa710553e7e74a5aaf51c1e9a5e2b9037772f38","observation_id":"65ee2097-0693-4d23-9a6e-91d17f2a279f","resolution":{"observed_at":"2026-05-21T23:25:45.286097Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.18982","last_updated":"2024-10-08T15:13:01Z","snapshot_observed_at":"2026-07-06T19:39:15.025945Z","submitted_at":"2024-10-08T15:13:01Z","title":"O1 Replication Journey: A Strategic Progress Report -- Part 1","version":1},"cited_work":{"arxiv_id":"2410.18982","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.18982","snapshot_observed_at":"2026-07-04T11:09:46.422756Z","title":"Ram´e, A.; Ferret, J.; Vieillard, N.; Dadashi, R.; Hussenot, L.; Cedoz, P.-L.; Sessa, P","venue":null,"work_id":"0738313f-d0cc-4548-a3bc-eba0fa2d607b","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2410.18982","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:27a33590fe0abebb713a145918a04cb13003a05bd2e442d988629f4f5b6aaef7","observation_id":"8d9cd287-48e2-4169-96c8-fb4ab79daf80","resolution":{"observed_at":"2026-05-21T23:25:45.349208Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16768","last_updated":"2024-06-24T16:24:34Z","snapshot_observed_at":"2026-07-06T18:36:07.983955Z","submitted_at":"2024-06-24T16:24:34Z","title":"WARP: On the Benefits of Weight Averaged Rewarded Policies","version":1},"cited_work":{"arxiv_id":"2406.16768","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.16768","snapshot_observed_at":"2026-07-04T10:49:45.640428Z","title":"Warp: On the benefits of weight averaged rewarded policies.arXiv preprint arXiv:2406.16768, 2024","venue":null,"work_id":"e22b5849-1e5c-4d18-9973-5ea982b594ef","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2406.16768","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:1b21bd5fad28742d5111e24a4af0d1fe7c55b57e37caf7072caa92abc094de02","observation_id":"62c27199-f8e2-4e43-9735-7098902b0d17","resolution":{"observed_at":"2026-05-21T23:25:45.301158Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:482acc5e9531db2550b1bf2505655e1c7acd1c274fa787be8bae2c708bcb960c","observation_id":"c3d41963-09c7-4cd6-a177-9b69d02f6db5","resolution":{"observed_at":"2026-05-21T23:25:45.313527Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05199","last_updated":"2023-11-29T14:45:53Z","snapshot_observed_at":"2026-07-06T16:29:27.783330Z","submitted_at":"2023-10-08T15:14:39Z","title":"Loose lips sink ships: Mitigating Length Bias in Reinforcement Learning from Human Feedback","version":5},"cited_work":{"arxiv_id":"2310.05199","doi":"10.48550/arxiv.2310.05199","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.05199","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Findings of the Association for Computational Linguistics: EMNLP , pages =","venue":"arXiv (Cornell University)","work_id":"c4bce86e-e2cb-4afe-bfee-9239fc94ad93","year":2023},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2310.05199","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:d076e712d9ed43e92b8b9501bb9fcfc39ce82129e59c6605bc4438f903839a0e","observation_id":"46083a4f-feed-45d9-bf47-7d29457ad3d1","resolution":{"observed_at":"2026-05-21T23:25:45.328850Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03716","last_updated":"2024-07-10T23:15:49Z","snapshot_observed_at":"2026-08-03T14:09:43.107865Z","submitted_at":"2023-10-05T17:38:28Z","title":"A Long Way to Go: Investigating Length Correlations in RLHF","version":2},"cited_work":{"arxiv_id":"2310.03716","doi":"10.48550/arxiv.2310.03716","metadata_source":"pith","pith_arxiv_id":"2310.03716","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A long way to go: Investigating length correlations in rlhf","venue":"cs.CL","work_id":"23d0dfe6-c919-4498-8696-1e298c7834e9","year":2023},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2310.03716","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:b143bf155d18ec86f4d24cc51d404d63f9d0543904e68b87c5b9bcf758802e56","observation_id":"24e63f3c-ad87-44a4-b51f-3942077965e3","resolution":{"observed_at":"2026-05-21T23:25:45.297132Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:49:21.130019+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:49:21.130019+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":"2408.03314","doi":"10.18653/v1/2025.acl-long.1486","metadata_source":"pith","pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","venue":"cs.LG","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:f35c0ee736a5f238e68032474e471386296f001dd42e8f81b03d8cfe8ac0fdf2","observation_id":"fedc1ac9-4c04-441a-bf70-3ec5c4844a97","resolution":{"observed_at":"2026-05-21T23:25:45.360482Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.09472","last_updated":"2024-12-10T08:54:09Z","snapshot_observed_at":"2026-07-06T17:44:37.894023Z","submitted_at":"2024-03-14T15:12:38Z","title":"Easy-to-Hard Generalization: Scalable Alignment Beyond Human Supervision","version":2},"cited_work":{"arxiv_id":"2403.09472","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.09472","snapshot_observed_at":"2026-07-01T20:56:14.059065Z","title":"Advances in Neural Information Processing Systems (NeurIPS) , year=","venue":null,"work_id":"87cc775f-7250-4dea-b110-cde2246d75a7","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2403.09472","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:cb9dd7b6deade9fe0e63761ade21f44445efee5afa1abfad0f7053e002c8d155","observation_id":"86982e4b-f595-4084-a8d2-944086c00017","resolution":{"observed_at":"2026-05-21T23:25:45.309813Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":"2408.00724","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-07-04T11:09:46.442555Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","venue":"cs.AI","work_id":"cef2c407-5d51-46a9-8eb6-f382b419502e","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:da30024a3fe9edfe39985d030790f4688b122a24227783650be58f26f0c62479","observation_id":"e484e6dd-d2f0-4ee9-a0c9-fcdc4b3c3939","resolution":{"observed_at":"2026-05-21T23:25:45.356641Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05692","last_updated":"2025-01-14T05:39:40Z","snapshot_observed_at":"2026-07-06T17:57:19.117933Z","submitted_at":"2024-04-08T17:18:04Z","title":"Evaluating Mathematical Reasoning Beyond Accuracy","version":2},"cited_work":{"arxiv_id":"2404.05692","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05692","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluating mathematical reasoning beyond accuracy","venue":null,"work_id":"b2ea2f90-9303-45bd-b0f3-f69dd12ff165","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2404.05692","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:2d449bd9fa3dcc96d009e912a3c762da1c00ce8dd8902abb37279b048711237d","observation_id":"099e8384-6906-4034-833d-cc9237fa9108","resolution":{"observed_at":"2026-05-21T23:25:45.320305Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:60dfde22c0b001c9a67de43965ef7703d003327bde418178e63807334a3ce7f0","observation_id":"8c9865bc-fc99-4cf6-956d-e6d378c1c85c","resolution":{"observed_at":"2026-05-21T23:25:45.264818Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15240","last_updated":"2025-02-22T10:21:46Z","snapshot_observed_at":"2026-07-06T19:06:41.697671Z","submitted_at":"2024-08-27T17:57:45Z","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","version":3},"cited_work":{"arxiv_id":"2408.15240","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.15240","snapshot_observed_at":"2026-07-08T21:25:38.629808Z","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","venue":"cs.LG","work_id":"dbe9c116-8030-4e7c-a259-b52c80846d0e","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2408.15240","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:e2e60853c9b8da2ce6091cae03969cf6e9dde411b2e65ad94a6fecd6d7f56b47","observation_id":"965e8885-c52a-4ce9-961b-875a5e9e0d61","resolution":{"observed_at":"2026-05-21T23:25:45.290017Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00891","last_updated":"2025-04-05T03:04:37Z","snapshot_observed_at":"2026-07-06T21:02:28.100677Z","submitted_at":"2025-04-01T15:21:05Z","title":"GenPRM: Scaling Test-Time Compute of Process Reward Models via Generative Reasoning","version":2},"cited_work":{"arxiv_id":"2504.00891","doi":"10.48550/arxiv.2504.00891","metadata_source":"arxiv_reference","pith_arxiv_id":"2504.00891","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2504.00891 , year=","venue":"ArXiv.org","work_id":"987b931c-ea64-480c-a5b5-e33b60f1c6b2","year":2025},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2504.00891","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:de11a8900e66edd8ec94faaf30d130d1b4222ac4ab8dba0cbb4eedb8c823bbe2","observation_id":"0e0c21f7-4a31-4d49-aa00-3063fb45931f","resolution":{"observed_at":"2026-05-21T23:25:45.293816Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06559","last_updated":"2025-05-26T14:03:32Z","snapshot_observed_at":"2026-08-05T20:30:49.812919Z","submitted_at":"2024-12-09T15:11:40Z","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","version":4},"cited_work":{"arxiv_id":"2412.06559","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.06559","snapshot_observed_at":"2026-07-04T08:39:42.307124Z","title":"Processbench: Identifying process errors in mathematical reasoning","venue":null,"work_id":"fd09974e-3188-49d4-a208-329b8acfbb53","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2412.06559","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:07aca10732cbc84dc1cd13850274818b41d62e2e6e7500e8735174e56b0cd05f","observation_id":"b93e8691-1a74-4db4-ae01-bb89c48410e0","resolution":{"observed_at":"2026-05-21T23:25:45.344436Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14361","last_updated":"2025-02-20T08:40:09Z","snapshot_observed_at":"2026-07-06T20:39:37.922156Z","submitted_at":"2025-02-20T08:40:09Z","title":"Retrieval-Augmented Process Reward Model for Generalizable Mathematical Reasoning","version":1},"cited_work":{"arxiv_id":"2502.14361","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14361","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2502.14361","venue":null,"work_id":"b6ac677e-aef7-40dd-9afc-72c3d8f85569","year":null},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2502.14361","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:3041bb4f8f638e45849301566c22c95b02247362f46b5bca0d20ed8f7f768b16","observation_id":"9ebc0511-b75b-49f4-8847-cc68dee08fd6","resolution":{"observed_at":"2026-05-21T23:25:45.275361Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11931","last_updated":"2024-06-17T13:51:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T13:51:35Z","title":"DeepSeek-Coder-V2: Breaking the Barrier of Closed-Source Models in Code Intelligence","version":1},"cited_work":{"arxiv_id":"2406.11931","doi":"10.48550/arxiv.2406.11931","metadata_source":"pith","pith_arxiv_id":"2406.11931","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-Coder-V2: Breaking the Barrier of Closed-Source Models in Code Intelligence","venue":"cs.SE","work_id":"713a39be-8c4e-4ee3-8671-11cf9379262f","year":2024},"citing_paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-21T23:24:43.556606Z"},"links":{"cited_paper":"/paper/2406.11931","citing_paper":"/paper/2507.15698"},"observation_digest":"sha256:0cdedce65f9eaeda31d948ac8bbf1d0ca43df805a3da6e0f184d43bb0ed56543","observation_id":"7b0b7362-6f40-4600-af13-74dd7cca7d5c","resolution":{"observed_at":"2026-05-21T23:25:45.304827Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.15698","last_updated":"2026-05-19T07:11:43Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-21T15:07:59Z","title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning"},"reference_resolution":{"displayed":29,"state_counts":{"malformed_identifier":0,"metadata_mismatch":15,"parse_uncertain":0,"unresolved":0,"verified_exact":14,"verified_fuzzy":0},"total_outbound_references":29},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 29 of 29 outbound references and 3 inbound Pith citation observations for arXiv:2507.15698."}