{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:U5QKEWGYUHTMU7KFUBYBVPQANF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"254a8a25b69b453df9a92a0ebd280e443b6a45319a97a9c07ee78dcd409f83d1","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-10-02T11:42:17Z","title_canon_sha256":"1d6fab20e5a2bcd12af53fe737f051d1a0b4a8d0d5530bd2d9346ac7f06de1b4"},"schema_version":"1.0","source":{"id":"2510.01925","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2510.01925","created_at":"2026-08-04T02:08:33Z"},{"alias_kind":"arxiv_version","alias_value":"2510.01925v3","created_at":"2026-08-04T02:08:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.01925","created_at":"2026-08-04T02:08:33Z"},{"alias_kind":"pith_short_12","alias_value":"U5QKEWGYUHTM","created_at":"2026-08-04T02:08:33Z"},{"alias_kind":"pith_short_16","alias_value":"U5QKEWGYUHTMU7KF","created_at":"2026-08-04T02:08:33Z"},{"alias_kind":"pith_short_8","alias_value":"U5QKEWGY","created_at":"2026-08-04T02:08:33Z"}],"graph_snapshots":[{"event_id":"sha256:c0f2be35ec894f3a04a244bbf027b94eea5f573e6b5c137fc3c3a4d8fa257557","target":"graph","created_at":"2026-08-04T02:08:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2510.01925/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward models (RMs) play a critical role in enhancing the reasoning performance of LLMs. For example, they can provide training signals to finetune LLMs during reinforcement learning (RL) and help select the best answer from multiple candidates during inference. In this paper, we provide a systematic introduction to RMs, along with a comprehensive survey of their applications in LLM reasoning. We first review fundamental concepts of RMs, including their architectures, training methodologies, and evaluation techniques. Then, we explore their key applications: (1) guiding generation and selectin","authors_text":"Hao Xu, Ning Miao, Qiyuan Liu, Wei Chen, Xuhong Chen, Yee Whye Teh","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-10-02T11:42:17Z","title":"Enhancing Large Language Model Reasoning with Reward Models: An Analytical Survey"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.01925","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:33a245f9ae26013e4f91cc189af41a11e1a6d2e081ecdeeebbd362f8a5b84943","target":"record","created_at":"2026-08-04T02:08:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"254a8a25b69b453df9a92a0ebd280e443b6a45319a97a9c07ee78dcd409f83d1","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-10-02T11:42:17Z","title_canon_sha256":"1d6fab20e5a2bcd12af53fe737f051d1a0b4a8d0d5530bd2d9346ac7f06de1b4"},"schema_version":"1.0","source":{"id":"2510.01925","kind":"arxiv","version":3}},"canonical_sha256":"a760a258d8a1e6ca7d45a0701abe0069515a4dafad579eea7c7dc4197b1b3f83","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a760a258d8a1e6ca7d45a0701abe0069515a4dafad579eea7c7dc4197b1b3f83","first_computed_at":"2026-08-04T02:08:33.612313Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-08-04T02:08:33.612313Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"elCUaHqBqpKE6Xx59alagIIlHT20azMi8t7glIfza3QVBx9noGRuKd1DINo2hUacXWn/L4KCidzLm+2WE8z5Dg==","signature_status":"signed_v1","signed_at":"2026-08-04T02:08:33.613873Z","signed_message":"canonical_sha256_bytes"},"source_id":"2510.01925","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:33a245f9ae26013e4f91cc189af41a11e1a6d2e081ecdeeebbd362f8a5b84943","sha256:c0f2be35ec894f3a04a244bbf027b94eea5f573e6b5c137fc3c3a4d8fa257557"],"state_sha256":"0e5c64f8c784c776f24809aeb1e6055828baf8c8f82ee7432137c02d4ac978c0"}