{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RXULUJPS6GQ6CUEWF6PRJPGFBM","short_pith_number":"pith:RXULUJPS","schema_version":"1.0","canonical_sha256":"8de8ba25f2f1a1e150962f9f14bcc50b1203c2bc59a9de11cc5ccadbd5da2e5c","source":{"kind":"arxiv","id":"2311.09724","version":2},"attestation_state":"computed","paper":{"title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Anningzhe Gao, Benyou Wang, Fei Yu","submitted_at":"2023-11-16T09:56:28Z","abstract_excerpt":"Large language models (LLMs) often struggle with maintaining accuracy throughout multiple multiple reasoning steps, especially in mathematical reasoning where an error in earlier steps can propagate to subsequent ones and it ultimately leading to an incorrect answer. To reduce error propagation, guided decoding is employed to direct the LM decoding on a step-by-step basis. We argue that in guided decoding, assessing the potential of an incomplete reasoning path can be more advantageous than simply ensuring per-step correctness, as the former approach leads towards a correct final answer. This "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09724","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-11-16T09:56:28Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"896ce8af59c09653325ef7951dc9400951153581cd7f4bb7472fd7a895490c3a","abstract_canon_sha256":"5a01fc8fd798908588dc1d03d0103069ff4ec39af3a21bd4e02b4304097c1ec1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:48.150707Z","signature_b64":"Ke5iXA1BMHRZTEHLmiJW7rGNvOu5JIxezgKKZDV+7Inf/MCYxgKbDlUQykUSyp6DngXG7E/cLCnWergh14oxDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8de8ba25f2f1a1e150962f9f14bcc50b1203c2bc59a9de11cc5ccadbd5da2e5c","last_reissued_at":"2026-07-05T08:02:48.150220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:48.150220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OVM, Outcome-supervised Value Models for Planning in Mathematical Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Anningzhe Gao, Benyou Wang, Fei Yu","submitted_at":"2023-11-16T09:56:28Z","abstract_excerpt":"Large language models (LLMs) often struggle with maintaining accuracy throughout multiple multiple reasoning steps, especially in mathematical reasoning where an error in earlier steps can propagate to subsequent ones and it ultimately leading to an incorrect answer. To reduce error propagation, guided decoding is employed to direct the LM decoding on a step-by-step basis. We argue that in guided decoding, assessing the potential of an incomplete reasoning path can be more advantageous than simply ensuring per-step correctness, as the former approach leads towards a correct final answer. This "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09724","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09724/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09724","created_at":"2026-07-05T08:02:48.150278+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09724v2","created_at":"2026-07-05T08:02:48.150278+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09724","created_at":"2026-07-05T08:02:48.150278+00:00"},{"alias_kind":"pith_short_12","alias_value":"RXULUJPS6GQ6","created_at":"2026-07-05T08:02:48.150278+00:00"},{"alias_kind":"pith_short_16","alias_value":"RXULUJPS6GQ6CUEW","created_at":"2026-07-05T08:02:48.150278+00:00"},{"alias_kind":"pith_short_8","alias_value":"RXULUJPS","created_at":"2026-07-05T08:02:48.150278+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.08146","citing_title":"Rewarding Progress: Scaling Automated Process Verifiers for LLM Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16302","citing_title":"Reducing Credit Assignment Variance via Counterfactual Reasoning Paths","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.14868","citing_title":"Goldilocks RL: Tuning Task Difficulty to Escape Sparse Rewards for Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16335","citing_title":"Beyond Verifiable Rewards: Rubric-Based GRM for Reinforced Fine-Tuning SWE Agents","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2312.08935","citing_title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06592","citing_title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23333","citing_title":"Process Supervision of Confidence Margin for Calibrated LLM Reasoning","ref_index":82,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM","json":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM.json","graph_json":"https://pith.science/api/pith-number/RXULUJPS6GQ6CUEWF6PRJPGFBM/graph.json","events_json":"https://pith.science/api/pith-number/RXULUJPS6GQ6CUEWF6PRJPGFBM/events.json","paper":"https://pith.science/paper/RXULUJPS"},"agent_actions":{"view_html":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM","download_json":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM.json","view_paper":"https://pith.science/paper/RXULUJPS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09724&json=true","fetch_graph":"https://pith.science/api/pith-number/RXULUJPS6GQ6CUEWF6PRJPGFBM/graph.json","fetch_events":"https://pith.science/api/pith-number/RXULUJPS6GQ6CUEWF6PRJPGFBM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM/action/storage_attestation","attest_author":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM/action/author_attestation","sign_citation":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM/action/citation_signature","submit_replication":"https://pith.science/pith/RXULUJPS6GQ6CUEWF6PRJPGFBM/action/replication_record"}},"created_at":"2026-07-05T08:02:48.150278+00:00","updated_at":"2026-07-05T08:02:48.150278+00:00"}