{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BNDGZ3FZMNENXFJLKN3OWE5AMQ","short_pith_number":"pith:BNDGZ3FZ","schema_version":"1.0","canonical_sha256":"0b466cecb96348db952b5376eb13a0640cb4db09621647c2d46b2049d9f82686","source":{"kind":"arxiv","id":"2412.01981","version":1},"attestation_state":"computed","paper":{"title":"Free Process Rewards without Process Labels","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bowen Zhou, Ganqu Cui, Hao Peng, Huayu Chen, Kaiyan Zhang, Lifan Yuan, Ning Ding, Wendi Li, Zhiyuan Liu","submitted_at":"2024-12-02T21:20:02Z","abstract_excerpt":"Different from its counterpart outcome reward models (ORMs), which evaluate the entire responses, a process reward model (PRM) scores a reasoning trajectory step by step, providing denser and more fine grained rewards. However, training a PRM requires labels annotated at every intermediate step, presenting significant challenges for both manual and automatic data collection. This paper aims to address this challenge. Both theoretically and empirically, we show that an \\textit{implicit PRM} can be obtained at no additional cost, by simply training an ORM on the cheaper response-level labels. Th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.01981","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-02T21:20:02Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"918e238225c16ae03b223a38ae4846370cba4ec8e73b17ae2d9dbc3c75d7a19a","abstract_canon_sha256":"380c643f8f08279065583c85910f6c37009177b63b2a3243b046865661cff6a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:43:41.877468Z","signature_b64":"Ms4xXctxzYpxXhsIWAOLHEeKDOtMOmdGPcocsNJ9R47iJdWIGYCV3vbD9r16bRmTQIIPrgTDcKssQsd3+GQ4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b466cecb96348db952b5376eb13a0640cb4db09621647c2d46b2049d9f82686","last_reissued_at":"2026-07-05T09:43:41.877063Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:43:41.877063Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Free Process Rewards without Process Labels","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bowen Zhou, Ganqu Cui, Hao Peng, Huayu Chen, Kaiyan Zhang, Lifan Yuan, Ning Ding, Wendi Li, Zhiyuan Liu","submitted_at":"2024-12-02T21:20:02Z","abstract_excerpt":"Different from its counterpart outcome reward models (ORMs), which evaluate the entire responses, a process reward model (PRM) scores a reasoning trajectory step by step, providing denser and more fine grained rewards. However, training a PRM requires labels annotated at every intermediate step, presenting significant challenges for both manual and automatic data collection. This paper aims to address this challenge. Both theoretically and empirically, we show that an \\textit{implicit PRM} can be obtained at no additional cost, by simply training an ORM on the cheaper response-level labels. Th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.01981","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.01981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.01981","created_at":"2026-07-05T09:43:41.877117+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.01981v1","created_at":"2026-07-05T09:43:41.877117+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.01981","created_at":"2026-07-05T09:43:41.877117+00:00"},{"alias_kind":"pith_short_12","alias_value":"BNDGZ3FZMNEN","created_at":"2026-07-05T09:43:41.877117+00:00"},{"alias_kind":"pith_short_16","alias_value":"BNDGZ3FZMNENXFJL","created_at":"2026-07-05T09:43:41.877117+00:00"},{"alias_kind":"pith_short_8","alias_value":"BNDGZ3FZ","created_at":"2026-07-05T09:43:41.877117+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07435","citing_title":"RLVP: Penalize the Path, Reward the Outcome","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01223","citing_title":"Theoria: Rewrite-Acceptability Verification over Informal Reasoning States","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03660","citing_title":"From Answers to States: Verifiable Process-Level Evaluation of Chemical Reasoning in Large Language Models","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01682","citing_title":"Off-the-Shelf LLMs as Process Scorers: Training-Free Alternative to PRMs for Mathematical Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14636","citing_title":"Teaching Large Language Models When Not to Know: Learning Temporal Critique for Ex-Ante Reasoning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07832","citing_title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21467","citing_title":"DelTA: Discriminative Token Credit Assignment for Reinforcement Learning from Verifiable Rewards","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11505","citing_title":"Selective Off-Policy Reference Tuning with Plan Guidance","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11505","citing_title":"Selective Off-Policy Reference Tuning with Plan Guidance","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11613","citing_title":"From Generic Correlation to Input-Specific Credit in On-Policy Self Distillation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01456","citing_title":"Process Reinforcement through Implicit Rewards","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21223","citing_title":"Zero-Shot Detection of LLM-Generated Text via Implicit Reward Model","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ","json":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ.json","graph_json":"https://pith.science/api/pith-number/BNDGZ3FZMNENXFJLKN3OWE5AMQ/graph.json","events_json":"https://pith.science/api/pith-number/BNDGZ3FZMNENXFJLKN3OWE5AMQ/events.json","paper":"https://pith.science/paper/BNDGZ3FZ"},"agent_actions":{"view_html":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ","download_json":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ.json","view_paper":"https://pith.science/paper/BNDGZ3FZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.01981&json=true","fetch_graph":"https://pith.science/api/pith-number/BNDGZ3FZMNENXFJLKN3OWE5AMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/BNDGZ3FZMNENXFJLKN3OWE5AMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ/action/storage_attestation","attest_author":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ/action/author_attestation","sign_citation":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ/action/citation_signature","submit_replication":"https://pith.science/pith/BNDGZ3FZMNENXFJLKN3OWE5AMQ/action/replication_record"}},"created_at":"2026-07-05T09:43:41.877117+00:00","updated_at":"2026-07-05T09:43:41.877117+00:00"}