{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZUQLYRYLSQDOXWUHY77NQQYHRA","short_pith_number":"pith:ZUQLYRYL","schema_version":"1.0","canonical_sha256":"cd20bc470b9406ebda87c7fed843078821bda247fd61cc574f384ceeca3082bc","source":{"kind":"arxiv","id":"2506.22157","version":1},"attestation_state":"computed","paper":{"title":"Training Language Model to Critique for Better Refinement","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bosi Wen, Chao Xiang, Chuxiong Sun, Cunxiang Wang, Jiale Cheng, Li Zhang, Mingchuan Yang, Minlie Huang, Pei Ke, Tianshu Yu, Xinyu Mu","submitted_at":"2025-06-27T12:10:57Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable evaluation and critique capabilities, providing insightful feedback and identifying flaws in various tasks. However, limited research has explored which types of critiques are most effective for improving model responses or how to generate such critiques. To address this gap, we introduce \\textbf{R}efinement-oriented \\textbf{C}ritique \\textbf{O}ptimization (RCO), a novel framework designed to train critic models using refinement signals. RCO uses a feedback loop where critiques, generated by the critic model, guide the actor model in re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.22157","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-27T12:10:57Z","cross_cats_sorted":[],"title_canon_sha256":"90368e8b96c7f88ad0823ec74baf0e2e3b0806bf2dd49f86c3426f935bc8d1e9","abstract_canon_sha256":"c47e1dba7305e1f9f6a472d7998d3d37974e876f52e93ff9bd498d1d811ebddf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:22.860597Z","signature_b64":"rdcgKYG4a7sXSlxukR2+ljDRlqhfJPiuouq/hu7AucyN4L8YaIuGyLrGye/OZ5N0qYUcBjXOyNlvsEYMlrwqDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd20bc470b9406ebda87c7fed843078821bda247fd61cc574f384ceeca3082bc","last_reissued_at":"2026-07-05T11:28:22.860122Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:22.860122Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Language Model to Critique for Better Refinement","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bosi Wen, Chao Xiang, Chuxiong Sun, Cunxiang Wang, Jiale Cheng, Li Zhang, Mingchuan Yang, Minlie Huang, Pei Ke, Tianshu Yu, Xinyu Mu","submitted_at":"2025-06-27T12:10:57Z","abstract_excerpt":"Large language models (LLMs) have demonstrated remarkable evaluation and critique capabilities, providing insightful feedback and identifying flaws in various tasks. However, limited research has explored which types of critiques are most effective for improving model responses or how to generate such critiques. To address this gap, we introduce \\textbf{R}efinement-oriented \\textbf{C}ritique \\textbf{O}ptimization (RCO), a novel framework designed to train critic models using refinement signals. RCO uses a feedback loop where critiques, generated by the critic model, guide the actor model in re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22157","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.22157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.22157","created_at":"2026-07-05T11:28:22.860189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.22157v1","created_at":"2026-07-05T11:28:22.860189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22157","created_at":"2026-07-05T11:28:22.860189+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZUQLYRYLSQDO","created_at":"2026-07-05T11:28:22.860189+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZUQLYRYLSQDOXWUH","created_at":"2026-07-05T11:28:22.860189+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZUQLYRYL","created_at":"2026-07-05T11:28:22.860189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.06794","citing_title":"No More Stale Feedback: Co-Evolving Critics for Open-World Agent Learning","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA","json":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA.json","graph_json":"https://pith.science/api/pith-number/ZUQLYRYLSQDOXWUHY77NQQYHRA/graph.json","events_json":"https://pith.science/api/pith-number/ZUQLYRYLSQDOXWUHY77NQQYHRA/events.json","paper":"https://pith.science/paper/ZUQLYRYL"},"agent_actions":{"view_html":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA","download_json":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA.json","view_paper":"https://pith.science/paper/ZUQLYRYL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.22157&json=true","fetch_graph":"https://pith.science/api/pith-number/ZUQLYRYLSQDOXWUHY77NQQYHRA/graph.json","fetch_events":"https://pith.science/api/pith-number/ZUQLYRYLSQDOXWUHY77NQQYHRA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA/action/storage_attestation","attest_author":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA/action/author_attestation","sign_citation":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA/action/citation_signature","submit_replication":"https://pith.science/pith/ZUQLYRYLSQDOXWUHY77NQQYHRA/action/replication_record"}},"created_at":"2026-07-05T11:28:22.860189+00:00","updated_at":"2026-07-05T11:28:22.860189+00:00"}