{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6P7VNA25W4FGIAJV2NASYNRHGM","short_pith_number":"pith:6P7VNA25","schema_version":"1.0","canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","source":{"kind":"arxiv","id":"2310.10080","version":1},"attestation_state":"computed","paper":{"title":"Let's reward step by step: Step-Level reward model as the Navigators for Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haotian Zhou, Hongxia Yang, Jianbo Yuan, Pengfei Liu, Qianli Ma, Tingkai Liu, Yang You","submitted_at":"2023-10-16T05:21:50Z","abstract_excerpt":"Recent years have seen considerable advancements in multi-step reasoning with Large Language Models (LLMs). The previous studies have elucidated the merits of integrating feedback or search mechanisms during model inference to improve the reasoning accuracy. The Process-Supervised Reward Model (PRM), typically furnishes LLMs with step-by-step feedback during the training phase, akin to Proximal Policy Optimization (PPO) or reject sampling. Our objective is to examine the efficacy of PRM in the inference phase to help discern the optimal solution paths for multi-step tasks such as mathematical "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10080","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:21:50Z","cross_cats_sorted":[],"title_canon_sha256":"15272afc4334d72abe4aabe7ebc8149d0b135c8ccb248fb68afb191bd06aa101","abstract_canon_sha256":"7fefbbf505a6d483f28f1ed988857e172d4bd9d67f68da505ccd1976ca925b9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:14.992217Z","signature_b64":"oFt0ZHr2FVBcS3T/F+XNf6smZkzB7nM6i9uG6RU3RlCQAU/xXGfnBTpT8RSKr6fIhgkj3OUI4P+yzcjcFs+PBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","last_reissued_at":"2026-07-05T07:01:14.991750Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:14.991750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Let's reward step by step: Step-Level reward model as the Navigators for Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haotian Zhou, Hongxia Yang, Jianbo Yuan, Pengfei Liu, Qianli Ma, Tingkai Liu, Yang You","submitted_at":"2023-10-16T05:21:50Z","abstract_excerpt":"Recent years have seen considerable advancements in multi-step reasoning with Large Language Models (LLMs). The previous studies have elucidated the merits of integrating feedback or search mechanisms during model inference to improve the reasoning accuracy. The Process-Supervised Reward Model (PRM), typically furnishes LLMs with step-by-step feedback during the training phase, akin to Proximal Policy Optimization (PPO) or reject sampling. Our objective is to examine the efficacy of PRM in the inference phase to help discern the optimal solution paths for multi-step tasks such as mathematical "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10080","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10080","created_at":"2026-07-05T07:01:14.991807+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10080v1","created_at":"2026-07-05T07:01:14.991807+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10080","created_at":"2026-07-05T07:01:14.991807+00:00"},{"alias_kind":"pith_short_12","alias_value":"6P7VNA25W4FG","created_at":"2026-07-05T07:01:14.991807+00:00"},{"alias_kind":"pith_short_16","alias_value":"6P7VNA25W4FGIAJV","created_at":"2026-07-05T07:01:14.991807+00:00"},{"alias_kind":"pith_short_8","alias_value":"6P7VNA25","created_at":"2026-07-05T07:01:14.991807+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05863","citing_title":"Strategic Bargaining in Multi-Buyer Markets: Reinforcement Learning from Verifiable Rewards for LLM Negotiations","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19808","citing_title":"Think Again or Think Longer? Selective Verification for Budget-Aware Reasoning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07367","citing_title":"Self-evolving LLM agents with in-distribution Optimization","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24635","citing_title":"HiMed: Incentivizing Hindi Reasoning in Medical LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30712","citing_title":"ExpGraph: Model-Agnostic Experience Learning with Graph-Structured Memory for LLM Agents","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07832","citing_title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2410.08146","citing_title":"Rewarding Progress: Scaling Automated Process Verifiers for LLM Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17877","citing_title":"PAIR: Prefix-Aware Internal Reward Model for Multi-Turn Agent Optimization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15529","citing_title":"Process Rewards with Learned Reliability","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14703","citing_title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2312.08935","citing_title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27977","citing_title":"SARL: Label-Free Reinforcement Learning by Rewarding Reasoning Topology","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06592","citing_title":"Improve Mathematical Reasoning in Language Models by Automated Process Supervision","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.21187","citing_title":"Do NOT Think That Much for 2+3=? On the Overthinking of o1-Like LLMs","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23366","citing_title":"GSAR: Typed Grounding for Hallucination Detection and Recovery in Multi-Agent LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04064","citing_title":"Improving Medical VQA through Trajectory-Aware Process Supervision","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM","json":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM.json","graph_json":"https://pith.science/api/pith-number/6P7VNA25W4FGIAJV2NASYNRHGM/graph.json","events_json":"https://pith.science/api/pith-number/6P7VNA25W4FGIAJV2NASYNRHGM/events.json","paper":"https://pith.science/paper/6P7VNA25"},"agent_actions":{"view_html":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM","download_json":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM.json","view_paper":"https://pith.science/paper/6P7VNA25","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10080&json=true","fetch_graph":"https://pith.science/api/pith-number/6P7VNA25W4FGIAJV2NASYNRHGM/graph.json","fetch_events":"https://pith.science/api/pith-number/6P7VNA25W4FGIAJV2NASYNRHGM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/action/storage_attestation","attest_author":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/action/author_attestation","sign_citation":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/action/citation_signature","submit_replication":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/action/replication_record"}},"created_at":"2026-07-05T07:01:14.991807+00:00","updated_at":"2026-07-05T07:01:14.991807+00:00"}