{"as_of":"2026-08-07T13:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:64f4584b89c9e6da63cbd43cd323aaf1db578a5610888e2fc5c4ab3088b9fbb5","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:55:00.569399Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T20:40:07.852377Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":137,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:0a048375405203aa8366497b5278547680ac3f8cc9a4b5ed5d384f506913e4f1","observation_id":"f95ff518-c0a2-4cf2-ba9b-6f309bdde87b","resolution":{"observed_at":"2026-05-12T08:40:42.043925Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-06T21:55:00.569399Z","title":"Process reward models for LLM agents: Practical framework and directions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23127","last_updated":"2025-06-29T07:31:24Z","snapshot_observed_at":"2026-08-07T10:28:04.865350Z","submitted_at":"2025-06-29T07:31:24Z","title":"Unleashing Embodied Task Planning Ability in LLMs via Reinforcement Learning","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T21:55:00.569399Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2506.23127"},"observation_digest":"sha256:ea993ea8d2213fd16f54d0ee1cf55995a022db4bce0be2b90578a11dabdfcd53","observation_id":"46e7171b-ec46-4bee-b76d-f05f0060db78","resolution":{"observed_at":"2026-08-06T21:55:00.569399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2507.14200","last_updated":"2026-05-15T13:08:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-14T16:17:11Z","title":"A Scalable Multi-LLM Collaboration System with Retrieval-based Selection and Exploration-Exploitation-Driven Enhancement","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-21T23:26:38.457193Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2507.14200"},"observation_digest":"sha256:4a48afa1f109bf656ea1d1b7fb5a2f85938d283a37dcebaa6b4773337718e585","observation_id":"7b677e12-6bf5-45d4-a241-5e8c246a3fcc","resolution":{"observed_at":"2026-05-21T23:30:46.105396Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":235,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:f8b5250d6a8f8a273a7fdc100839d8d84d6684e213960cd3ee7854397dc5cc3c","observation_id":"e9162390-ac2b-46fe-8592-48cd973764e3","resolution":{"observed_at":"2026-05-14T22:23:15.209112Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-05T15:44:14.788155Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.19598","last_updated":"2025-08-27T06:19:50Z","snapshot_observed_at":"2026-08-05T15:44:13.449330Z","submitted_at":"2025-08-27T06:19:50Z","title":"Encouraging Good Processes Without the Need for Good Answers: Reinforcement Learning for LLM Agent Planning","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T15:44:14.788155Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2508.19598"},"observation_digest":"sha256:65a3d1584cefd1c1746bfa0de5dbe061ab1b41d401dfc78769e822c3c008c1cf","observation_id":"4baf07d6-19cf-4135-a5b3-e240d8b1c703","resolution":{"observed_at":"2026-08-05T15:44:14.788155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-05T12:24:02.368310Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.01684","last_updated":"2025-09-01T18:04:10Z","snapshot_observed_at":"2026-08-05T12:24:00.598679Z","submitted_at":"2025-09-01T18:04:10Z","title":"Reinforcement Learning for Machine Learning Engineering Agents","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-05T12:24:02.368310Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2509.01684"},"observation_digest":"sha256:cd6f4623aae428270eb513e197f6863a2091934aebe3303b65e92f486c647857","observation_id":"ef0ee11a-e246-440f-8cbe-5c3556f6446b","resolution":{"observed_at":"2026-08-05T12:24:02.368310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-08-03T09:07:42.489237Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"reference_index":268,"source":"pdf_text","source_observed_at":"2026-05-18T19:19:36.427337Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2509.02547"},"observation_digest":"sha256:56f4b06c3549e531def478fd4971a474bfc77c9e23b6d5164aee4753c6cf3d02","observation_id":"c4d0eb90-6505-4219-a8fb-e67540e346a8","resolution":{"observed_at":"2026-05-18T19:21:48.656981Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-04T07:56:46.023170Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.24803","last_updated":"2026-07-15T00:10:20Z","snapshot_observed_at":"2026-08-05T09:19:56.278937Z","submitted_at":"2025-10-28T00:48:20Z","title":"MASPRM: Multi-Agent System Process Reward Model","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-04T07:56:46.023170Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2510.24803"},"observation_digest":"sha256:748aecb8bc5c222bdc9af3ccfdbc2dc0b3b2d0ef0a62456fa280a601e3fba38c","observation_id":"69ce9a63-a132-439c-b4ef-6b5382d7eb0c","resolution":{"observed_at":"2026-08-04T07:56:46.023170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-03T03:04:43.473321Z","title":"Process reward models for llm agents: Practical framework and directions.arXiv preprint arXiv:2502.10325,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.09305","last_updated":"2026-06-28T18:13:37Z","snapshot_observed_at":"2026-08-03T07:10:48.024369Z","submitted_at":"2026-02-10T00:45:24Z","title":"Reward Modeling for Reinforcement Learning-Based LLM Reasoning: Design, Challenges, and Evaluation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T03:04:43.473321Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2602.09305"},"observation_digest":"sha256:c44ffd2cfd3f1656f40ee9e73155755cf335f8dfccbed2f92e1eb4c8159aabcf","observation_id":"8c44698b-6b8a-4b98-94a4-386534cb4aa6","resolution":{"observed_at":"2026-08-03T03:04:43.473321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2604.07851","last_updated":"2026-04-09T06:07:03Z","snapshot_observed_at":"2026-08-03T12:28:38.894496Z","submitted_at":"2026-04-09T06:07:03Z","title":"ReRec: Reasoning-Augmented LLM-based Recommendation Assistant via Reinforcement Fine-tuning","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T17:29:34.145855Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2604.07851"},"observation_digest":"sha256:1b4b1fd951996d82f9d6bd4a24d3c5903c14a373a3e25040104977a162f1d19b","observation_id":"735e5750-b183-46b6-acf9-150fa5b7b843","resolution":{"observed_at":"2026-05-11T06:41:38.272747Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2605.06200","last_updated":"2026-05-07T13:09:31Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T13:09:31Z","title":"A$^2$TGPO: Agentic Turn-Group Policy Optimization with Adaptive Turn-level Clipping","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-08T10:41:46.675257Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2605.06200"},"observation_digest":"sha256:41ca2e0d662ad3fb657f81e04425fa1ec1f21051569fbe614036c6392ceab5c2","observation_id":"76ecb7bf-6c00-4e45-b94c-db02d10d2c44","resolution":{"observed_at":"2026-05-11T19:56:08.764793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2605.09934","last_updated":"2026-05-11T03:32:55Z","snapshot_observed_at":"2026-08-03T16:23:18.233908Z","submitted_at":"2026-05-11T03:32:55Z","title":"TRACER: Verifiable Generative Provenance for Multimodal Tool-Using Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-12T04:30:06.869916Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2605.09934"},"observation_digest":"sha256:ec5b9049bf80026103e901b33ac3ffb09a77a0321bb0fba8d279bac449260db5","observation_id":"d1017cb9-f713-412d-af1d-b9b8e08afcd1","resolution":{"observed_at":"2026-05-12T06:11:26.376883Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2605.11235","last_updated":"2026-05-11T20:50:29Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:50:29Z","title":"Internalizing Curriculum Judgment for LLM Reinforcement Fine-Tuning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T02:19:27.345348Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2605.11235"},"observation_digest":"sha256:5fc66ddc54a00d8a5318a0f86c8a8d4fbf0ca607fc38f063742f0f2fb7a2c2a6","observation_id":"ab9ed4e5-0ccb-48b8-8d12-bc2b2b370c58","resolution":{"observed_at":"2026-05-13T02:22:06.790866Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.07027","last_updated":"2026-06-05T08:17:28Z","snapshot_observed_at":"2026-08-01T08:01:00.499508Z","submitted_at":"2026-06-05T08:17:28Z","title":"StainFlow: Entity-Stain Tracking and Evidence Linking for Process Rewards in GUI Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T22:24:41.353303Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.07027"},"observation_digest":"sha256:5ec4a7665a17f17e09a6c50f431d3f8c8c9d321d822f3714f6f1bc603718bf25","observation_id":"64ff4a60-ed75-4796-8d76-ca22865b8c53","resolution":{"observed_at":"2026-07-02T16:47:09.644058Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.07367","last_updated":"2026-06-05T15:09:52Z","snapshot_observed_at":"2026-08-04T14:57:01.386223Z","submitted_at":"2026-06-05T15:09:52Z","title":"Self-evolving LLM agents with in-distribution Optimization","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-27T22:18:27.021136Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.07367"},"observation_digest":"sha256:ef8045a48ec9619b56a63754c897b991b1a22e11745ff6a27918bd2f5d73a129","observation_id":"750b729d-0b41-4b49-85f6-316fd64d974a","resolution":{"observed_at":"2026-07-02T16:57:09.563002Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.21262","last_updated":"2026-07-23T15:28:36Z","snapshot_observed_at":"2026-08-02T20:51:48.133765Z","submitted_at":"2026-06-19T09:38:02Z","title":"ARCO: Adaptive Rubrics with Co-Evolution for Multi-Step LLM-Based Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T14:33:50.123077Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.21262"},"observation_digest":"sha256:a75dba60f2f4d02d3edeb062626f4a08c029fc9553895876c437196a276b15d4","observation_id":"ab4bda4e-7106-4530-bc4d-d072e9e97af4","resolution":{"observed_at":"2026-07-04T06:19:38.638684Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-02T10:43:52.874792Z","title":"Process reward models for LLM agents: Practical framework and directions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.21262","last_updated":"2026-07-23T15:28:36Z","snapshot_observed_at":"2026-08-02T20:51:48.133765Z","submitted_at":"2026-06-19T09:38:02Z","title":"ARCO: Adaptive Rubrics with Co-Evolution for Multi-Step LLM-Based Agents","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-02T10:43:52.874792Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.21262"},"observation_digest":"sha256:e0788098526928208c997115accb387a4674d01b45f6faa411b2d69fd54ca0f5","observation_id":"2811bb84-3870-4bee-964b-335b758adde9","resolution":{"observed_at":"2026-08-02T10:43:52.874792Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.21399","last_updated":"2026-06-19T13:08:17Z","snapshot_observed_at":"2026-07-06T23:56:26.805329Z","submitted_at":"2026-06-19T13:08:17Z","title":"Calibration Is Not Control: Why LLM-Agent Oversight Needs Intervention","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T14:24:38.527103Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.21399"},"observation_digest":"sha256:1dd5187910d5283fd9e9bf4527d83474794cdfa862bbbb51aa15239ffab38051","observation_id":"24e18af3-2760-4054-af69-5b4f913f1edf","resolution":{"observed_at":"2026-07-04T06:29:37.873379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.23640","last_updated":"2026-06-22T17:30:24Z","snapshot_observed_at":"2026-08-02T09:22:48.099895Z","submitted_at":"2026-06-22T17:30:24Z","title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-26T09:20:35.062060Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.23640"},"observation_digest":"sha256:295376028872adf1adbe1a7f231c8ac4fe9d5e2afeedb903753deb6debfd1d4a","observation_id":"1dbb1f91-93d6-4b4a-a8a2-d9717bde3001","resolution":{"observed_at":"2026-07-04T09:59:44.572615Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.25852","last_updated":"2026-06-24T14:02:13Z","snapshot_observed_at":"2026-07-07T00:00:17.090702Z","submitted_at":"2026-06-24T14:02:13Z","title":"Semantic Consistency Policy Optimization for Reinforcement Learning of LLM Agents","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-25T20:16:39.347676Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.25852"},"observation_digest":"sha256:2f03c27ec336c2b87c7596e64e1a34a0dda715d63d0339d3adc93a62c41eae23","observation_id":"63338ea6-52ae-41cc-a02b-f373b9e1161b","resolution":{"observed_at":"2026-07-04T20:20:07.668163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.26080","last_updated":"2026-06-24T17:54:08Z","snapshot_observed_at":"2026-08-02T15:38:18.729836Z","submitted_at":"2026-06-24T17:54:08Z","title":"Neglected Free Lunch from Post-training: Progress Advantage for LLM Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-25T19:55:51.114244Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.26080"},"observation_digest":"sha256:aa4561231f66ce39ce7930f4f154f041fb12f21f11a3bd664a5cf51ac57231c2","observation_id":"4694c7c6-38ac-4a83-aab6-7911e90eeabb","resolution":{"observed_at":"2026-07-04T20:40:07.853911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":"2502.10325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-07-04T20:40:07.852377Z","title":"Process reward models for llm agents: Practical framework and directions","venue":null,"work_id":"7770478b-9aa2-4e18-a113-729fbe69a4de","year":2025},"citing_paper":{"arxiv_id":"2606.29745","last_updated":"2026-06-29T03:42:01Z","snapshot_observed_at":"2026-08-04T21:51:46.402646Z","submitted_at":"2026-06-29T03:42:01Z","title":"ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T04:36:35.213066Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2606.29745"},"observation_digest":"sha256:3142d6d9ebbdfcad9f117dc02a581b927bb659166218681c6f71ff19a0445d8c","observation_id":"c7939e92-aad6-47c9-bf80-6d2e901c8b1c","resolution":{"observed_at":"2026-06-30T16:44:56.609798Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.10325","snapshot_observed_at":"2026-08-01T18:53:08.965286Z","title":"arXiv:2502.10325 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17149","last_updated":"2026-07-19T09:14:27Z","snapshot_observed_at":"2026-08-05T07:57:20.874529Z","submitted_at":"2026-07-19T09:14:27Z","title":"A Diagnostic Framework for AI Agent Behavior","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-01T18:53:08.965286Z"},"links":{"cited_paper":"/paper/2502.10325","citing_paper":"/paper/2607.17149"},"observation_digest":"sha256:9c6bea80d4d2a9802e47e27fb94b286e2beaaaf571f7a6dffa1c3f65c7302159","observation_id":"fd198367-8d74-4d80-8bef-178f768b551c","resolution":{"observed_at":"2026-08-01T18:53:08.965286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.10325/citation-record","integrity":"/paper/2502.10325/integrity","json":"/paper/2502.10325/citation-record.json","paper":"/paper/2502.10325"},"outbound":[],"paper":{"arxiv_id":"2502.10325","last_updated":"2025-02-14T17:34:28Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T20:36:45.378436Z","submitted_at":"2025-02-14T17:34:28Z","title":"Process Reward Models for LLM Agents: Practical Framework and Directions"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2502.10325."}