{"as_of":"2026-08-07T23:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cbbe88d72f461717a9a87d93907094fa29eac60fbef2bef8503c67ebee494763","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:47:33.769274Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T13:20:32.432002Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T05:17:40.139105Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"cited_work":{"arxiv_id":"2507.15024","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.15024","snapshot_observed_at":"2026-07-03T05:17:40.139105Z","title":"arXiv preprint arXiv:2507.15024 , year =","venue":null,"work_id":"841334aa-828b-46b9-9df9-8d1240113be1","year":null},"citing_paper":{"arxiv_id":"2606.11078","last_updated":"2026-06-09T16:39:10Z","snapshot_observed_at":"2026-07-06T23:50:11.160843Z","submitted_at":"2026-06-09T16:39:10Z","title":"A History-Aware Visually Grounded Critic for Computer Use Agents","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-27T13:20:32.432002Z"},"links":{"cited_paper":"/paper/2507.15024","citing_paper":"/paper/2606.11078"},"observation_digest":"sha256:7edfbb4978e754aeb01fa3cf2002f2f1935a9fb104d52bdc52052d11c23a869d","observation_id":"e14f3855-713e-4386-ae59-1a6f75f5b28f","resolution":{"observed_at":"2026-07-03T05:17:40.140375Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.15024/citation-record","integrity":"/paper/2507.15024/integrity","json":"/paper/2507.15024/citation-record.json","paper":"/paper/2507.15024"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:30.497379Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.497379Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:7724df5faf43dac6abc3fadadcf3e3e76b420b8ea36b61d310ca246c31a9309e","observation_id":"4f90c26e-6e16-4087-8845-8faec0fc3134","resolution":{"observed_at":"2026-08-06T15:47:30.497379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:30.538772Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.538772Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:892d371c6fd918d74f1b07cbf25ce829055bfd4744e31ec3c2ed03bb0ac9d1f4","observation_id":"3591f7da-6e0f-40ca-b8ec-12e89ec7c928","resolution":{"observed_at":"2026-08-06T15:47:30.538772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11791","last_updated":"2024-08-21T17:24:15Z","snapshot_observed_at":"2026-07-06T19:04:09.389583Z","submitted_at":"2024-08-21T17:24:15Z","title":"Critique-out-Loud Reward Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.11791","snapshot_observed_at":"2026-08-06T15:47:30.588128Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.588128Z"},"links":{"cited_paper":"/paper/2408.11791","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:bd92477ed37506276b9cadd54322ee5b258cf607e34ea9e32904609bef807f84","observation_id":"6f7414c9-713f-489a-a2f9-ed4b1c21aeac","resolution":{"observed_at":"2026-08-06T15:47:30.588128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:35.391763Z","title":null,"venue":null,"work_id":"b337e802-6a02-4910-b82d-ba006c8a2efe","year":2005},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.626895Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:c1e038a3c606896964fd1ba33263f8609e8806f680643b8667802f065e88c06f","observation_id":"aa17fe7c-c25c-4f5f-917d-bd674ebb3b9c","resolution":{"observed_at":"2026-08-06T15:47:35.491866Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19162","last_updated":"2025-05-17T14:32:38Z","snapshot_observed_at":"2026-08-07T15:58:52.021141Z","submitted_at":"2025-04-27T08:45:06Z","title":"SPC: Evolving Self-Play Critic via Adversarial Games for LLM Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.19162","snapshot_observed_at":"2026-08-06T15:47:30.674164Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.674164Z"},"links":{"cited_paper":"/paper/2504.19162","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:0da73a4e07101cadf144577cf1295be00f91c73521a3d2c453d307685b2f4d71","observation_id":"449b5695-ac43-4588-9625-8b28fde75fd8","resolution":{"observed_at":"2026-08-06T15:47:30.674164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11140","last_updated":"2025-01-06T21:18:24Z","snapshot_observed_at":"2026-08-01T22:46:19.585574Z","submitted_at":"2024-02-17T00:13:36Z","title":"Boosting of Thoughts: Trial-and-Error Problem Solving with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11140","snapshot_observed_at":"2026-08-06T15:47:30.790393Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.790393Z"},"links":{"cited_paper":"/paper/2402.11140","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:bf5b663d1c83e57a94e246711990e67abcbe45f5e30e8daa09b097b049457d9c","observation_id":"4dc8b538-237c-412c-9832-37712ac325a3","resolution":{"observed_at":"2026-08-06T15:47:30.790393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-06T15:47:30.853821Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.853821Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:72bf129dd1af893857747f63d197e446ebc5f939302610d134366e23fd916f97","observation_id":"5fb05160-88d7-460a-91d7-1820eabc0d91","resolution":{"observed_at":"2026-08-06T15:47:30.853821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14008","last_updated":"2024-06-06T13:19:44Z","snapshot_observed_at":"2026-08-03T03:39:09.398343Z","submitted_at":"2024-02-21T18:49:26Z","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14008","snapshot_observed_at":"2026-08-06T15:47:30.972701Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:30.972701Z"},"links":{"cited_paper":"/paper/2402.14008","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:0d30c0bd3a3196dcb04756b37afd4be3745ac358970a2c06b66b498bb0c56291","observation_id":"db27df59-873d-4426-982d-71b0e9e39cd3","resolution":{"observed_at":"2026-08-06T15:47:30.972701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.19361","last_updated":"2025-03-30T14:48:59Z","snapshot_observed_at":"2026-08-07T17:45:25.804804Z","submitted_at":"2025-02-26T17:59:27Z","title":"Can Large Language Models Detect Errors in Long Chain-of-Thought Reasoning?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.19361","snapshot_observed_at":"2026-08-06T15:47:31.064312Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.064312Z"},"links":{"cited_paper":"/paper/2502.19361","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:2bf0854de683702112dd4cb5724000e28937b3d8c60465166f47cccaf1addf7b","observation_id":"e33a6f0a-07a2-4590-8043-8b9129eee944","resolution":{"observed_at":"2026-08-06T15:47:31.064312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12186","last_updated":"2024-11-12T13:24:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-18T17:57:57Z","title":"Qwen2.5-Coder Technical Report","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12186","snapshot_observed_at":"2026-08-06T15:47:31.143756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.143756Z"},"links":{"cited_paper":"/paper/2409.12186","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:a6951d5669cc9fbc498788aa57917bac19d3241b7811cdad66e6ed9ec380237a","observation_id":"99210596-09d1-48dd-8795-647088ace763","resolution":{"observed_at":"2026-08-06T15:47:31.143756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-06T15:47:31.229548Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.229548Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:2aaba87d3e2ae662bef05c1d767b37c35e1badae49d27e208b276f1786a36d05","observation_id":"3a3d8ee3-725a-4d21-be1e-510383e8109a","resolution":{"observed_at":"2026-08-06T15:47:31.229548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-08-06T15:47:31.341295Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.341295Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:bb1879f56e695d5ebce2456d75e18ebbc61f275e95008a41bc2fcc2f0ffe2bbb","observation_id":"c84370b2-5226-4678-baf5-c5b3e66f8c6c","resolution":{"observed_at":"2026-08-06T15:47:31.341295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:31.420751Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.420751Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:f48437d212664226ea2e70a35aaaf0e61637269148e78ab9b23baa4c3689700b","observation_id":"69fa5b1c-27e6-489f-a8fe-bd2913737502","resolution":{"observed_at":"2026-08-06T15:47:31.420751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:35.147702Z","title":null,"venue":null,"work_id":"cfd62c52-ede2-440b-bb52-463f416384a0","year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.510699Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:20054579f095596af4422c9c6a8bbc3d8cbd14e523827b93bdb1645079bbf43e","observation_id":"f3b07d0f-9fef-4dcc-bd58-03d9d0aded52","resolution":{"observed_at":"2026-08-06T15:47:35.271981Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:34.871087Z","title":null,"venue":null,"work_id":"227b78e9-c2be-4595-9030-7d37ae319b48","year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.620920Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:9eaeaa8cd0a86b2a9e78da533d7023aacdf119c669f6aea8217613e205bec8e3","observation_id":"7ecf92e4-9072-4f79-b15c-b1096c2eaa18","resolution":{"observed_at":"2026-08-06T15:47:34.996452Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-05T13:11:04.104454Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-06T15:47:31.690472Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.690472Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:20d0b14c8abc19e016752dbea640a744cbf4aba145757232047cb98def077893","observation_id":"1ddfd3e9-9406-46be-83f0-477ab0a1bfc5","resolution":{"observed_at":"2026-08-06T15:47:31.690472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14809","last_updated":"2024-06-01T07:46:28Z","snapshot_observed_at":"2026-07-06T17:34:11.950339Z","submitted_at":"2024-02-22T18:59:02Z","title":"CriticBench: Benchmarking LLMs for Critique-Correct Reasoning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14809","snapshot_observed_at":"2026-08-06T15:47:31.805450Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.805450Z"},"links":{"cited_paper":"/paper/2402.14809","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:91c496cd4d5466696c9721147e88e89ed9a9186aa216413555aaebcf5b962480","observation_id":"6d143611-5771-45c2-b162-1a9dcf1eea7c","resolution":{"observed_at":"2026-08-06T15:47:31.805450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:31.927640Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:31.927640Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:4a810c94cd6c5a226eb8e522205a1ffae5c7673bb2785f5253961dfba5cae7bb","observation_id":"aed862b1-c602-44a4-82f4-8ebca8615d41","resolution":{"observed_at":"2026-08-06T15:47:31.927640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12832","last_updated":"2024-10-02T17:58:39Z","snapshot_observed_at":"2026-08-03T10:06:52.418096Z","submitted_at":"2024-10-02T17:58:39Z","title":"Generative Reward Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12832","snapshot_observed_at":"2026-08-06T15:47:32.042776Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.042776Z"},"links":{"cited_paper":"/paper/2410.12832","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:5ee49c1f785b9ec0a7cad2a75eea149ea45a4014f5c2beae55d19530fa266b7c","observation_id":"6ce12686-ffac-4a00-95d6-eb6f36f0b990","resolution":{"observed_at":"2026-08-06T15:47:32.042776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00215","last_updated":"2024-06-28T19:53:17Z","snapshot_observed_at":"2026-07-06T18:38:46.314431Z","submitted_at":"2024-06-28T19:53:17Z","title":"LLM Critics Help Catch LLM Bugs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00215","snapshot_observed_at":"2026-08-06T15:47:32.141323Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.141323Z"},"links":{"cited_paper":"/paper/2407.00215","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:2b224d7ee8858732e9f8dbcee713fcc57ab77251114dfa9630fc28ba973a9e72","observation_id":"395ff2a7-3a94-42ad-afc1-0e52bc8d9e71","resolution":{"observed_at":"2026-08-06T15:47:32.141323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:32.251144Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.251144Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:89c63b324cc6796a420be2d726306c951add11842a805a8969b657d712ccb157","observation_id":"fe3c777f-a9a9-482b-8ac1-7bc38453e2d4","resolution":{"observed_at":"2026-08-06T15:47:32.251144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-06T15:47:32.354499Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.354499Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:7209ad466634b5d46e0d2f47ab2b7fc9c3a6fe8c6cf3e29f3691fc3871785db9","observation_id":"e32f7fd4-7b89-49a2-89d4-06f1fe121628","resolution":{"observed_at":"2026-08-06T15:47:32.354499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10337","last_updated":"2025-04-16T14:58:26Z","snapshot_observed_at":"2026-08-07T16:05:50.114818Z","submitted_at":"2025-04-14T15:46:33Z","title":"Heimdall: test-time scaling on the generative verification","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10337","snapshot_observed_at":"2026-08-06T15:47:32.464146Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.464146Z"},"links":{"cited_paper":"/paper/2504.10337","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:2632f9dd9472a4781f682a1aa79ad8bd2a21718649f1eb55e49fb03281a23503","observation_id":"512245a2-62d9-4b47-845f-d8d2968ee0fb","resolution":{"observed_at":"2026-08-06T15:47:32.464146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-06T15:47:32.548031Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.548031Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:8d07a03e7d65059d650588ff11d4a3334572caa404864c1d7c30418726f75f5b","observation_id":"cfd9916b-b335-47b2-a4d8-ec9c1f4fb40d","resolution":{"observed_at":"2026-08-06T15:47:32.548031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05727","last_updated":"2025-08-04T02:14:13Z","snapshot_observed_at":"2026-07-06T20:19:07.706492Z","submitted_at":"2025-01-10T05:51:52Z","title":"Self-Evolving Critique Abilities in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05727","snapshot_observed_at":"2026-08-06T15:47:32.650831Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.650831Z"},"links":{"cited_paper":"/paper/2501.05727","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:96add2e8bed2f0262186e78655a7dff8818c48a8f4e96f73eda5b43df3804cf3","observation_id":"1e1cb7b9-361f-423f-b192-1d6d3dcc962d","resolution":{"observed_at":"2026-08-06T15:47:32.650831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14492","last_updated":"2025-01-24T13:48:10Z","snapshot_observed_at":"2026-07-06T20:25:36.218904Z","submitted_at":"2025-01-24T13:48:10Z","title":"RealCritic: Towards Effectiveness-Driven Evaluation of Language Model Critiques","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14492","snapshot_observed_at":"2026-08-06T15:47:32.729477Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.729477Z"},"links":{"cited_paper":"/paper/2501.14492","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:cf67629dfe59cb3308ad73bffb2f360d17c698b68af2ec7ea3ba6024307835a9","observation_id":"d81f6338-9e43-4d73-bece-a415c274ad59","resolution":{"observed_at":"2026-08-06T15:47:32.729477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:32.817233Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.817233Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:7286f8cbdc14e833187688984083ea782b9cd3fd92da4e1b10ee81f37f35401f","observation_id":"899a2785-16bc-4678-b659-9736ec3457ce","resolution":{"observed_at":"2026-08-06T15:47:32.817233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.14275","last_updated":"2022-11-25T18:19:44Z","snapshot_observed_at":"2026-08-01T02:16:43.109337Z","submitted_at":"2022-11-25T18:19:44Z","title":"Solving math word problems with process- and outcome-based feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.14275","snapshot_observed_at":"2026-08-06T15:47:32.904058Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:32.904058Z"},"links":{"cited_paper":"/paper/2211.14275","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:0bed2cab55a58648df5cb089a299d7305bc34a1bff0c26e5de13c5b77e5c4512","observation_id":"cbfad0d3-5e83-4ea7-a463-291d5f253719","resolution":{"observed_at":"2026-08-06T15:47:32.904058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08935","last_updated":"2024-02-19T14:07:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-14T13:41:54Z","title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08935","snapshot_observed_at":"2026-08-06T15:47:33.007036Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.007036Z"},"links":{"cited_paper":"/paper/2312.08935","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:782199e63674cf2faf03ae95c702862c4100a8c5f335981210b39035279e41be","observation_id":"1f082e5a-6744-4873-b8e5-0330a6545774","resolution":{"observed_at":"2026-08-06T15:47:33.007036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00724","last_updated":"2025-03-03T07:53:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-01T17:16:04Z","title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00724","snapshot_observed_at":"2026-08-06T15:47:33.068552Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.068552Z"},"links":{"cited_paper":"/paper/2408.00724","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:3ee6c033cb40389fc5693a3234df7571d69033773cf4af0cf59f5ad0ff911318","observation_id":"81de2408-7c57-4eb4-a988-12b937d18fe3","resolution":{"observed_at":"2026-08-06T15:47:33.068552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T15:47:33.161410Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.161410Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:7c0133739f550511808d344d390e9ce1792d00539affd311c53f1f43572ba90f","observation_id":"91982f38-b9ba-4dad-9675-4ff95486070d","resolution":{"observed_at":"2026-08-06T15:47:33.161410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.00662","last_updated":"2025-05-01T17:03:17Z","snapshot_observed_at":"2026-08-07T15:57:19.356999Z","submitted_at":"2025-05-01T17:03:17Z","title":"DeepCritic: Deliberate Critique with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.00662","snapshot_observed_at":"2026-08-06T15:47:33.218455Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.218455Z"},"links":{"cited_paper":"/paper/2505.00662","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:fbbb1589caac728b1a50a73db8c371e7362a75e6c74edf7e0de5beedb2a4a842","observation_id":"294f5c9b-2f2c-4f5c-84ac-77f27ec6074b","resolution":{"observed_at":"2026-08-06T15:47:33.218455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:33.334831Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.334831Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:a470bacfbad1527844ed2edd52c1ab089115bd9eb91b840b6770dbf5329acfdd","observation_id":"b7c1ccc2-b88c-4715-a1ec-9f04bc5b4cb7","resolution":{"observed_at":"2026-08-06T15:47:33.334831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:34.579075Z","title":null,"venue":null,"work_id":"82bfea9a-300f-4b93-b376-5d09b7f0d2d3","year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.411742Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:8ae247cbb08ae6fa3abae7202ad8181aded3bda18ef9ccd114cad643b0f6c359","observation_id":"8d7389d0-9cca-4d9a-929a-fe97407a60de","resolution":{"observed_at":"2026-08-06T15:47:34.703734Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07301","last_updated":"2025-06-05T16:34:24Z","snapshot_observed_at":"2026-08-03T11:11:25.359494Z","submitted_at":"2025-01-13T13:10:16Z","title":"The Lessons of Developing Process Reward Models in Mathematical Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07301","snapshot_observed_at":"2026-08-06T15:47:33.515460Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.515460Z"},"links":{"cited_paper":"/paper/2501.07301","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:f97dd87cfe171cd300a24abd88e4d6d58aac69653641519e9501d8781d66060c","observation_id":"21395f0b-6e69-4a79-8e1f-e50c6f89ba4c","resolution":{"observed_at":"2026-08-06T15:47:33.515460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06559","last_updated":"2025-05-26T14:03:32Z","snapshot_observed_at":"2026-08-05T20:30:49.812919Z","submitted_at":"2024-12-09T15:11:40Z","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06559","snapshot_observed_at":"2026-08-06T15:47:33.591244Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.591244Z"},"links":{"cited_paper":"/paper/2412.06559","citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:dd88ca2cbf96190f3e157b927f4725105afe46b9674e0fbee3dace61c2035210","observation_id":"0d914c5f-701e-485f-a7f2-c7e916999e3c","resolution":{"observed_at":"2026-08-06T15:47:33.591244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:33.690552Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.690552Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:766bfcf01cdfef3f1ba5039df9c738cd724df1f29331d58c7e33b35d8c0d473a","observation_id":"77882341-f8f7-4bfe-adf2-7c2cf0f6da3b","resolution":{"observed_at":"2026-08-06T15:47:33.690552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T15:47:34.383505Z","title":null,"venue":null,"work_id":"4f6779fb-31ce-4439-a663-6dd560f61ccf","year":2025},"citing_paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T15:47:33.769274Z"},"links":{"citing_paper":"/paper/2507.15024"},"observation_digest":"sha256:6387506cf86f2286f68f9b243d74061dde9f1cf3f6ae75c74addf9c4bfb1638e","observation_id":"60bd0749-61dd-470d-80bf-726a22ebc83b","resolution":{"observed_at":"2026-08-06T15:47:34.475927Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.15024","last_updated":"2025-07-20T16:19:51Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T23:16:39.881011Z","submitted_at":"2025-07-20T16:19:51Z","title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 1 inbound Pith citation observation for arXiv:2507.15024."}