{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U5QKEWGYUHTMU7KFUBYBVPQANF","short_pith_number":"pith:U5QKEWGY","schema_version":"1.0","canonical_sha256":"a760a258d8a1e6ca7d45a0701abe0069515a4dafad579eea7c7dc4197b1b3f83","source":{"kind":"arxiv","id":"2510.01925","version":3},"attestation_state":"computed","paper":{"title":"Enhancing Large Language Model Reasoning with Reward Models: An Analytical Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Xu, Ning Miao, Qiyuan Liu, Wei Chen, Xuhong Chen, Yee Whye Teh","submitted_at":"2025-10-02T11:42:17Z","abstract_excerpt":"Reward models (RMs) play a critical role in enhancing the reasoning performance of LLMs. For example, they can provide training signals to finetune LLMs during reinforcement learning (RL) and help select the best answer from multiple candidates during inference. In this paper, we provide a systematic introduction to RMs, along with a comprehensive survey of their applications in LLM reasoning. We first review fundamental concepts of RMs, including their architectures, training methodologies, and evaluation techniques. Then, we explore their key applications: (1) guiding generation and selectin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.01925","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-10-02T11:42:17Z","cross_cats_sorted":[],"title_canon_sha256":"1d6fab20e5a2bcd12af53fe737f051d1a0b4a8d0d5530bd2d9346ac7f06de1b4","abstract_canon_sha256":"254a8a25b69b453df9a92a0ebd280e443b6a45319a97a9c07ee78dcd409f83d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:08:33.613873Z","signature_b64":"elCUaHqBqpKE6Xx59alagIIlHT20azMi8t7glIfza3QVBx9noGRuKd1DINo2hUacXWn/L4KCidzLm+2WE8z5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a760a258d8a1e6ca7d45a0701abe0069515a4dafad579eea7c7dc4197b1b3f83","last_reissued_at":"2026-08-04T02:08:33.612313Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:08:33.612313Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Large Language Model Reasoning with Reward Models: An Analytical Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Xu, Ning Miao, Qiyuan Liu, Wei Chen, Xuhong Chen, Yee Whye Teh","submitted_at":"2025-10-02T11:42:17Z","abstract_excerpt":"Reward models (RMs) play a critical role in enhancing the reasoning performance of LLMs. For example, they can provide training signals to finetune LLMs during reinforcement learning (RL) and help select the best answer from multiple candidates during inference. In this paper, we provide a systematic introduction to RMs, along with a comprehensive survey of their applications in LLM reasoning. We first review fundamental concepts of RMs, including their architectures, training methodologies, and evaluation techniques. Then, we explore their key applications: (1) guiding generation and selectin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.01925","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.01925/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.01925","created_at":"2026-08-04T02:08:33.613779+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.01925v3","created_at":"2026-08-04T02:08:33.613779+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.01925","created_at":"2026-08-04T02:08:33.613779+00:00"},{"alias_kind":"pith_short_12","alias_value":"U5QKEWGYUHTM","created_at":"2026-08-04T02:08:33.613779+00:00"},{"alias_kind":"pith_short_16","alias_value":"U5QKEWGYUHTMU7KF","created_at":"2026-08-04T02:08:33.613779+00:00"},{"alias_kind":"pith_short_8","alias_value":"U5QKEWGY","created_at":"2026-08-04T02:08:33.613779+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.29296","citing_title":"Process Advantage Signal Shaping: A Paradigm-Agnostic Middleware for Process-Supervised RL in LLM Reasoners","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":77,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF","json":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF.json","graph_json":"https://pith.science/api/pith-number/U5QKEWGYUHTMU7KFUBYBVPQANF/graph.json","events_json":"https://pith.science/api/pith-number/U5QKEWGYUHTMU7KFUBYBVPQANF/events.json","paper":"https://pith.science/paper/U5QKEWGY"},"agent_actions":{"view_html":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF","download_json":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF.json","view_paper":"https://pith.science/paper/U5QKEWGY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.01925&json=true","fetch_graph":"https://pith.science/api/pith-number/U5QKEWGYUHTMU7KFUBYBVPQANF/graph.json","fetch_events":"https://pith.science/api/pith-number/U5QKEWGYUHTMU7KFUBYBVPQANF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF/action/storage_attestation","attest_author":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF/action/author_attestation","sign_citation":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF/action/citation_signature","submit_replication":"https://pith.science/pith/U5QKEWGYUHTMU7KFUBYBVPQANF/action/replication_record"}},"created_at":"2026-08-04T02:08:33.613779+00:00","updated_at":"2026-08-04T02:08:33.613779+00:00"}