{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SRV56KWXDFPA2JJ7GJ2H32FPCY","short_pith_number":"pith:SRV56KWX","schema_version":"1.0","canonical_sha256":"946bdf2ad7195e0d253f32747de8af162b5d0f92e234b224953e7be32f5998fb","source":{"kind":"arxiv","id":"2410.11287","version":2},"attestation_state":"computed","paper":{"title":"Process Reward Model with Q-Value Rankings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Wendi Li, Yixuan Li","submitted_at":"2024-10-15T05:10:34Z","abstract_excerpt":"Process Reward Modeling (PRM) is critical for complex reasoning and decision-making tasks where the accuracy of intermediate steps significantly influences the overall outcome. Existing PRM approaches, primarily framed as classification problems, employ cross-entropy loss to independently evaluate each step's correctness. This method can lead to suboptimal reward distribution and does not adequately address the interdependencies among steps. To address these limitations, we introduce the Process Q-value Model (PQM), a novel framework that redefines PRM in the context of a Markov Decision Proce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11287","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-15T05:10:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"72ceb0857272f5e6693d2708bf17f95d49bdb83fc596517fc7d8dcc960e93698","abstract_canon_sha256":"9a46a7e2dd9da3a80da832e9e67c47a8550fc42f48a8a96b6e92db4ac47f51c4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:21.342312Z","signature_b64":"RQF2Jkea9ofD9lBBXBpowReEhbDcw5jsaCORPj7ZEd1Ao2r9jz6Big59c8XEKR27ssMzXYUaNof1GrdF/NsbBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"946bdf2ad7195e0d253f32747de8af162b5d0f92e234b224953e7be32f5998fb","last_reissued_at":"2026-07-05T10:12:21.340953Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:21.340953Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Process Reward Model with Q-Value Rankings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Wendi Li, Yixuan Li","submitted_at":"2024-10-15T05:10:34Z","abstract_excerpt":"Process Reward Modeling (PRM) is critical for complex reasoning and decision-making tasks where the accuracy of intermediate steps significantly influences the overall outcome. Existing PRM approaches, primarily framed as classification problems, employ cross-entropy loss to independently evaluate each step's correctness. This method can lead to suboptimal reward distribution and does not adequately address the interdependencies among steps. To address these limitations, we introduce the Process Q-value Model (PQM), a novel framework that redefines PRM in the context of a Markov Decision Proce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11287","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11287/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11287","created_at":"2026-07-05T10:12:21.341015+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11287v2","created_at":"2026-07-05T10:12:21.341015+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11287","created_at":"2026-07-05T10:12:21.341015+00:00"},{"alias_kind":"pith_short_12","alias_value":"SRV56KWXDFPA","created_at":"2026-07-05T10:12:21.341015+00:00"},{"alias_kind":"pith_short_16","alias_value":"SRV56KWXDFPA2JJ7","created_at":"2026-07-05T10:12:21.341015+00:00"},{"alias_kind":"pith_short_8","alias_value":"SRV56KWX","created_at":"2026-07-05T10:12:21.341015+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27369","citing_title":"Reinforcement Learning without Ground-Truth Solutions can Improve LLMs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15698","citing_title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2501.09732","citing_title":"Inference-Time Scaling for Diffusion Models beyond Scaling Denoising Steps","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11853","citing_title":"GEAR: Granularity-Adaptive Advantage Reweighting for LLM Agents via Self-Distillation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11853","citing_title":"GEAR: Granularity-Adaptive Advantage Reweighting for LLM Agents via Self-Distillation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21776","citing_title":"Video-R1: Reinforcing Video Reasoning in MLLMs","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY","json":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY.json","graph_json":"https://pith.science/api/pith-number/SRV56KWXDFPA2JJ7GJ2H32FPCY/graph.json","events_json":"https://pith.science/api/pith-number/SRV56KWXDFPA2JJ7GJ2H32FPCY/events.json","paper":"https://pith.science/paper/SRV56KWX"},"agent_actions":{"view_html":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY","download_json":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY.json","view_paper":"https://pith.science/paper/SRV56KWX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11287&json=true","fetch_graph":"https://pith.science/api/pith-number/SRV56KWXDFPA2JJ7GJ2H32FPCY/graph.json","fetch_events":"https://pith.science/api/pith-number/SRV56KWXDFPA2JJ7GJ2H32FPCY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY/action/storage_attestation","attest_author":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY/action/author_attestation","sign_citation":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY/action/citation_signature","submit_replication":"https://pith.science/pith/SRV56KWXDFPA2JJ7GJ2H32FPCY/action/replication_record"}},"created_at":"2026-07-05T10:12:21.341015+00:00","updated_at":"2026-07-05T10:12:21.341015+00:00"}