{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IZBLJPOAZQKCIQDVIR6QWBWNAK","short_pith_number":"pith:IZBLJPOA","schema_version":"1.0","canonical_sha256":"4642b4bdc0cc14244075447d0b06cd02aade8d0521263c803f76ea06707af44b","source":{"kind":"arxiv","id":"2507.15024","version":1},"attestation_state":"computed","paper":{"title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Yu, Hao Xiang, Hongyu Lin, Junyang Lin, Le Sun, Le Yu, Qiaoyu Tang, Xianpei Han, Yaojie Lu","submitted_at":"2025-07-20T16:19:51Z","abstract_excerpt":"With the rapid advancement of Large Language Models (LLMs), developing effective critic modules for precise guidance has become crucial yet challenging. In this paper, we initially demonstrate that supervised fine-tuning for building critic modules (which is widely adopted in current solutions) fails to genuinely enhance models' critique abilities, producing superficial critiques with insufficient reflections and verifications. To unlock the unprecedented critique capabilities, we propose RefCritic, a long-chain-of-thought critic module based on reinforcement learning with dual rule-based rewa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15024","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-20T16:19:51Z","cross_cats_sorted":[],"title_canon_sha256":"dd8325ea125ba48c2940605e6fff200a9abf3680227917f89511ca83ee69f953","abstract_canon_sha256":"e78e2a15cba9bfc622ccf346ccf5ffc66bd6d2f473199599ceb339db3aa0af48"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:17.014848Z","signature_b64":"VmnbibPpY+4rWcn/4MV8cgj2JgGXb/nCShB+A4qY4KU3LBMYI9dNdzMerzJOebPueOdw8rnjUtgEntH4tfy2Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4642b4bdc0cc14244075447d0b06cd02aade8d0521263c803f76ea06707af44b","last_reissued_at":"2026-07-05T11:40:17.014100Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:17.014100Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RefCritic: Training Long Chain-of-Thought Critic Models with Refinement Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bowen Yu, Hao Xiang, Hongyu Lin, Junyang Lin, Le Sun, Le Yu, Qiaoyu Tang, Xianpei Han, Yaojie Lu","submitted_at":"2025-07-20T16:19:51Z","abstract_excerpt":"With the rapid advancement of Large Language Models (LLMs), developing effective critic modules for precise guidance has become crucial yet challenging. In this paper, we initially demonstrate that supervised fine-tuning for building critic modules (which is widely adopted in current solutions) fails to genuinely enhance models' critique abilities, producing superficial critiques with insufficient reflections and verifications. To unlock the unprecedented critique capabilities, we propose RefCritic, a long-chain-of-thought critic module based on reinforcement learning with dual rule-based rewa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15024","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15024/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15024","created_at":"2026-07-05T11:40:17.014196+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15024v1","created_at":"2026-07-05T11:40:17.014196+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15024","created_at":"2026-07-05T11:40:17.014196+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZBLJPOAZQKC","created_at":"2026-07-05T11:40:17.014196+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZBLJPOAZQKCIQDV","created_at":"2026-07-05T11:40:17.014196+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZBLJPOA","created_at":"2026-07-05T11:40:17.014196+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11078","citing_title":"A History-Aware Visually Grounded Critic for Computer Use Agents","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK","json":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK.json","graph_json":"https://pith.science/api/pith-number/IZBLJPOAZQKCIQDVIR6QWBWNAK/graph.json","events_json":"https://pith.science/api/pith-number/IZBLJPOAZQKCIQDVIR6QWBWNAK/events.json","paper":"https://pith.science/paper/IZBLJPOA"},"agent_actions":{"view_html":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK","download_json":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK.json","view_paper":"https://pith.science/paper/IZBLJPOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15024&json=true","fetch_graph":"https://pith.science/api/pith-number/IZBLJPOAZQKCIQDVIR6QWBWNAK/graph.json","fetch_events":"https://pith.science/api/pith-number/IZBLJPOAZQKCIQDVIR6QWBWNAK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK/action/storage_attestation","attest_author":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK/action/author_attestation","sign_citation":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK/action/citation_signature","submit_replication":"https://pith.science/pith/IZBLJPOAZQKCIQDVIR6QWBWNAK/action/replication_record"}},"created_at":"2026-07-05T11:40:17.014196+00:00","updated_at":"2026-07-05T11:40:17.014196+00:00"}