{"as_of":"2026-08-06T14:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fb88fb1a1f2336c2010e65e0d8156782464d9f984f77cee41418b0deaa283c8e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T04:30:45.669787Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2510.14703","last_updated":"2026-04-28T18:17:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-16T14:06:03Z","title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-18T06:30:39.858246Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2510.14703"},"observation_digest":"sha256:9768569294fdf7ceea71716dd32a688d33b257133ef23fed2368d3223518b1f8","observation_id":"2d850bde-0282-4b18-a290-f3dc3a01aeff","resolution":{"observed_at":"2026-05-18T06:30:59.555716Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-02T19:57:33.949237Z","title":"Zheng et al","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.06652","last_updated":"2026-06-11T03:22:36Z","snapshot_observed_at":"2026-08-02T19:57:28.702185Z","submitted_at":"2026-02-28T04:33:11Z","title":"PaLMR: Towards Faithful Visual Reasoning via Multimodal Process Alignment","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-02T19:57:33.949237Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2603.06652"},"observation_digest":"sha256:3413173b52ab9bf1bc492e671b22b78163f2646043976ebaa4cb65369b805ced","observation_id":"6a81ab97-54aa-4679-9cc2-b1894b2689f6","resolution":{"observed_at":"2026-08-02T19:57:33.949237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.07941","last_updated":"2026-04-16T04:43:04Z","snapshot_observed_at":"2026-08-02T07:24:04.075021Z","submitted_at":"2026-04-09T08:00:37Z","title":"Large Language Model Post-Training: A Unified View of Off-Policy and On-Policy Learning","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T18:28:58.515666Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.07941"},"observation_digest":"sha256:242d02e9ddc750616ef56346d18e060e4dd60217088d17cea2d0c4f502d8b0aa","observation_id":"f309b32f-ba66-4810-b73b-d98d9592556a","resolution":{"observed_at":"2026-05-11T00:30:53.448267Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.08224","last_updated":"2026-04-09T13:19:41Z","snapshot_observed_at":"2026-08-02T22:56:00.067549Z","submitted_at":"2026-04-09T13:19:41Z","title":"Externalization in LLM Agents: A Unified Review of Memory, Skills, Protocols and Harness Engineering","version":1},"reference_index":196,"source":"arxiv_source","source_observed_at":"2026-05-10T17:40:14.733882Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.08224"},"observation_digest":"sha256:97f9399d48bd7da6e32380df516e5ca91334dd253af4d8baf143030ff8897bc3","observation_id":"5fc80b76-4701-4c90-9dd6-7e8aca71f89c","resolution":{"observed_at":"2026-05-11T06:20:59.523271Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.13602","last_updated":"2026-04-15T08:11:34Z","snapshot_observed_at":"2026-07-06T23:01:32.542040Z","submitted_at":"2026-04-15T08:11:34Z","title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","version":1},"reference_index":126,"source":"pdf_text","source_observed_at":"2026-05-10T13:58:53.430492Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.13602"},"observation_digest":"sha256:bfdd89febdc2ed18b51a6d7655f02e7de54f36eabb24e81701224a5665b28796","observation_id":"932b2a9a-9e3c-4721-8568-38eda0d226e2","resolution":{"observed_at":"2026-05-10T14:00:28.516058Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.23333","last_updated":"2026-04-25T14:40:13Z","snapshot_observed_at":"2026-07-06T23:09:34.300163Z","submitted_at":"2026-04-25T14:40:13Z","title":"Process Supervision of Confidence Margin for Calibrated LLM Reasoning","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-08T08:19:09.437464Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.23333"},"observation_digest":"sha256:b4fffa4d7381b88373b6aa757a346b9bb11276063b6e52a0bfc4fee3beb96a16","observation_id":"5fdcce4e-9326-46b4-bab9-8b0875ca0526","resolution":{"observed_at":"2026-05-11T20:41:12.237411Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.23809","last_updated":"2026-04-26T17:13:17Z","snapshot_observed_at":"2026-07-06T23:09:58.193241Z","submitted_at":"2026-04-26T17:13:17Z","title":"LegalDrill: Diagnosis-Driven Synthesis for Legal Reasoning in Small Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-08T06:06:53.959061Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.23809"},"observation_digest":"sha256:f26d8aa3cb9e026bff96c53e111110f092a6553880bdc1cd3de1d942cc1a7c7f","observation_id":"434f1bb5-6ed0-4fca-b823-fca0e54f6391","resolution":{"observed_at":"2026-05-11T21:16:25.962737Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.24198","last_updated":"2026-06-20T11:27:04Z","snapshot_observed_at":"2026-08-02T16:58:42.709299Z","submitted_at":"2026-04-27T09:00:30Z","title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-08T03:47:34.897401Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.24198"},"observation_digest":"sha256:c8422fed0cdcd5e2dd17ccfc4f23040b3bac40bdd34db23e8911b3f0e309a617","observation_id":"1ed8a920-2437-4abc-a660-a8e17e2a1942","resolution":{"observed_at":"2026-05-09T00:19:25.993336Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.24198","last_updated":"2026-06-20T11:27:04Z","snapshot_observed_at":"2026-08-02T16:58:42.709299Z","submitted_at":"2026-04-27T09:00:30Z","title":"Rewarding the Scientific Process: Process-Level Reward Modeling for Agentic Data Analysis","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-07-01T09:13:20.071265Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.24198"},"observation_digest":"sha256:4a7531a3ce935a8b3bee857f35396ce26ff274a207862cdfa9edc0cc1952bd5f","observation_id":"d58d4ea6-070a-4d41-826e-141868c78f62","resolution":{"observed_at":"2026-07-01T09:15:42.558536Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2604.24583","last_updated":"2026-04-27T15:08:02Z","snapshot_observed_at":"2026-07-06T23:10:33.821313Z","submitted_at":"2026-04-27T15:08:02Z","title":"Improving Vision-language Models with Perception-centric Process Reward Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-08T04:33:36.634359Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2604.24583"},"observation_digest":"sha256:9cf5a6b16cc859c2d834f79ea90170353a6cef418e059ac8aab3158d678a8be1","observation_id":"6d71e468-2da3-4ea2-9322-7474b7952a3b","resolution":{"observed_at":"2026-05-11T21:41:16.840951Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2605.02819","last_updated":"2026-05-04T16:56:01Z","snapshot_observed_at":"2026-07-06T23:15:51.008483Z","submitted_at":"2026-05-04T16:56:01Z","title":"SCPRM: A Schema-aware Cumulative Process Reward Model for Knowledge Graph Question Answering","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-08T17:57:18.130401Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2605.02819"},"observation_digest":"sha256:003da4c158ebb9436f63ced19bb7b6d7620d5326b4416c0949afc52a8a85d246","observation_id":"59e34ba7-0921-45af-bf08-f04947d5aacd","resolution":{"observed_at":"2026-05-09T06:55:43.621498Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2605.15951","last_updated":"2026-05-15T13:41:41Z","snapshot_observed_at":"2026-07-06T23:27:11.118592Z","submitted_at":"2026-05-15T13:41:41Z","title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-05-20T18:39:11.904941Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2605.15951"},"observation_digest":"sha256:7c9420ca893b90a0fe6893815ca8eafe8475294a3027a63825c94854a973108a","observation_id":"4b36b1c9-172a-4812-8301-95552cf2bfd8","resolution":{"observed_at":"2026-05-20T18:43:38.769945Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2606.01436","last_updated":"2026-05-31T20:15:12Z","snapshot_observed_at":"2026-08-01T20:14:15.209379Z","submitted_at":"2026-05-31T20:15:12Z","title":"Learning from Saturated Data: Signals Beyond Correctness for LLM Training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T17:10:05.592596Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2606.01436"},"observation_digest":"sha256:9eb44494b445073b1a2fa83187d06f6c86bf98a372b70080b28af127b8203c49","observation_id":"ccb30e36-f67b-4ef9-9dcf-5c0cc37f6ded","resolution":{"observed_at":"2026-06-28T17:12:24.561892Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2606.17735","last_updated":"2026-06-16T09:55:45Z","snapshot_observed_at":"2026-08-06T04:38:28.462018Z","submitted_at":"2026-06-16T09:55:45Z","title":"Shattering the Autoregressive Curse: Dynamic Epistemic Entropy Orchestrated Erasable Reinforcement Learning for LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T00:48:56.892634Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2606.17735"},"observation_digest":"sha256:649ad321a9ef59f90529f506a8b8debad7b3643b4c34dfcdd8ea3850a628a8dd","observation_id":"ea62d952-3925-4f51-9dc8-d2fe78c4813d","resolution":{"observed_at":"2026-07-03T21:18:58.377454Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":"2510.08049","doi":"10.48550/arxiv.2510.08049","metadata_source":"pith","pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","venue":"cs.CL","work_id":"1b364adb-2d12-4e61-89fc-804d607b2735","year":2025},"citing_paper":{"arxiv_id":"2606.19893","last_updated":"2026-06-18T07:50:40Z","snapshot_observed_at":"2026-08-03T01:32:15.946410Z","submitted_at":"2026-06-18T07:50:40Z","title":"MetaResearcher: Scaling Deep Research via Self-Reflective Reinforcement Learning in Adversarial Virtual Environments","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-26T17:43:03.092837Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2606.19893"},"observation_digest":"sha256:15b01ff5a8a0cdb3cc561943bd8a5a38893783c26db0506dc661eb5c318a289b","observation_id":"49477570-1edd-4303-b655-6eca31098101","resolution":{"observed_at":"2026-07-04T03:39:31.181688Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-07-13T05:02:55.608400Z","title":"A survey of process reward models: From outcome signals to process supervisions for large language models.arXiv preprint arXiv:2510.08049, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.09153","last_updated":"2026-07-10T07:16:43Z","snapshot_observed_at":"2026-08-05T15:37:38.707202Z","submitted_at":"2026-07-10T07:16:43Z","title":"KV-PRM: Efficient Process Reward Modeling via KV-Cache Transfer for Multi-Agent Test-Time Scaling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-13T05:02:55.608400Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2607.09153"},"observation_digest":"sha256:992fb532c4cfaea95de180035d6b431f3e175656e98a19787e5f8b65b913b5ed","observation_id":"4945d923-54f8-4128-837e-3e623c309379","resolution":{"observed_at":"2026-07-13T05:02:55.608400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-07-14T15:43:24.809948Z","title":"Zheng, J","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.09790","last_updated":"2026-07-08T16:54:30Z","snapshot_observed_at":"2026-08-05T11:41:24.758599Z","submitted_at":"2026-07-08T16:54:30Z","title":"Semantic Drift and the Stability of Operator Control in Reasoning-Class Decision Support Systems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-14T15:43:24.809948Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2607.09790"},"observation_digest":"sha256:b074d3bbc07dc4122763ec98a98d3bfe6348a56d44cb5b833573723e94d5bc9c","observation_id":"69719b0b-ac22-4bba-9b34-be23546ea564","resolution":{"observed_at":"2026-07-14T15:43:24.809948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-01T13:31:02.965705Z","title":"CoRR , volume =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19450","last_updated":"2026-07-29T12:12:34Z","snapshot_observed_at":"2026-08-05T06:59:22.496239Z","submitted_at":"2026-07-21T13:36:23Z","title":"REGEN: Replay-recycling for Expert-to-Generalist distillation with Offline Reinforcement Learning","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-01T13:31:02.965705Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2607.19450"},"observation_digest":"sha256:dfb99b1f35eb102b239ae308bb045a5de3ee1e9bec9692f4ef53c95aa36a37aa","observation_id":"e4aefbf9-c8c4-4576-afdb-3f49d95be509","resolution":{"observed_at":"2026-08-01T13:31:02.965705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-03T16:55:02.896387Z","title":"arXiv preprint arXiv:2510.08049 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28942","last_updated":"2026-08-03T20:04:55Z","snapshot_observed_at":"2026-08-06T14:27:42.302439Z","submitted_at":"2026-07-31T01:53:41Z","title":"NeSyFS: A Neuro-symbolic Fast-Slow Thinking Framework for LLM Agent under Partial Observability","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-03T16:55:02.896387Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2607.28942"},"observation_digest":"sha256:bf18c0b8677850b77d3d349b336aa324afd2ac1b3d1a8595fa332074ad7d2bba","observation_id":"c8ce994c-2139-4f6a-a705-8a1eebf1e7ce","resolution":{"observed_at":"2026-08-03T16:55:02.896387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-05T04:24:56.708873Z","title":"arXiv preprint arXiv:2510.08049 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28942","last_updated":"2026-08-03T20:04:55Z","snapshot_observed_at":"2026-08-06T14:27:42.302439Z","submitted_at":"2026-07-31T01:53:41Z","title":"NeSyFS: A Neuro-symbolic Fast-Slow Thinking Framework for LLM Agent under Partial Observability","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-05T04:24:56.708873Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2607.28942"},"observation_digest":"sha256:56da4dbf219d73960799cf29594444d11e7e281eb1a912e39075c26fab7112c5","observation_id":"f8900606-bcef-415b-ab27-00dbdcdac4e5","resolution":{"observed_at":"2026-08-05T04:24:56.708873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.08049","snapshot_observed_at":"2026-08-06T04:30:45.669787Z","title":"A survey of process reward models: From outcome signals to process supervisions for large language models.arXiv preprint arXiv:2510.08049, 2025","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05139","last_updated":"2026-08-05T17:57:16Z","snapshot_observed_at":"2026-08-06T14:40:27.175286Z","submitted_at":"2026-08-05T17:57:16Z","title":"Toward Skill-Native LLMs: Skill Entropy for Benchmarking and Training Long-Horizon Reasoning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T04:30:45.669787Z"},"links":{"cited_paper":"/paper/2510.08049","citing_paper":"/paper/2608.05139"},"observation_digest":"sha256:55848d9cf917fd16d53dfe234a0bcd1dba3fc33e02eda8baf1f840d658ec0ea6","observation_id":"51c4fc5c-8ba0-4182-8d2b-6574abb25afd","resolution":{"observed_at":"2026-08-06T04:30:45.669787Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2510.08049/citation-record","integrity":"/paper/2510.08049/integrity","json":"/paper/2510.08049/citation-record.json","paper":"/paper/2510.08049"},"outbound":[],"paper":{"arxiv_id":"2510.08049","last_updated":"2026-04-29T02:51:24Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T22:32:07.945004Z","submitted_at":"2025-10-09T10:35:31Z","title":"A Survey of Process Reward Models: From Outcome Signals to Process Supervisions for Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2510.08049."}