{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZGXPLEYEGD7RFJFDZ2UFCCRAMV","short_pith_number":"pith:ZGXPLEYE","schema_version":"1.0","canonical_sha256":"c9aef5930430ff12a4a3cea8510a20656b8712636b9d73137463cfb487f2d069","source":{"kind":"arxiv","id":"2501.03124","version":5},"attestation_state":"computed","paper":{"title":"PRMBench: A Fine-grained and Challenging Benchmark for Process-Level Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiawei Zhou, Mingyang Song, Xiaoye Qu, Yu Cheng, Zhaochen Su","submitted_at":"2025-01-06T16:31:45Z","abstract_excerpt":"Process-level Reward Models (PRMs) are crucial for complex reasoning and decision-making tasks, where each intermediate step plays an important role in the reasoning process. Since language models are prone to various types of errors during the reasoning process, PRMs are required to possess nuanced capabilities for detecting various implicit error types in real-world scenarios. However, current benchmarks primarily focus on step correctness, failing to evaluate PRMs' performance systematically. To address this gap, we introduce PRMBench, a process-level benchmark specifically designed to asse"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.03124","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-06T16:31:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f9d73ad938a531752f3dba936d52f90121e4e948b7d3c9469711c2c13c68ee29","abstract_canon_sha256":"68c37ebb1dd5735f5986d4b1842284fbc0acad002fe35675d5554a07c154a2a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:33.281986Z","signature_b64":"Gpa79jHaQsPrDZ5taboEHl1LJU3PfL3lzcifncUc/ahwn+E8/lykhAz3t8NVw7oZUXU7A4RzdmPMOymOWbyGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9aef5930430ff12a4a3cea8510a20656b8712636b9d73137463cfb487f2d069","last_reissued_at":"2026-07-05T11:28:33.281468Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:33.281468Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PRMBench: A Fine-grained and Challenging Benchmark for Process-Level Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiawei Zhou, Mingyang Song, Xiaoye Qu, Yu Cheng, Zhaochen Su","submitted_at":"2025-01-06T16:31:45Z","abstract_excerpt":"Process-level Reward Models (PRMs) are crucial for complex reasoning and decision-making tasks, where each intermediate step plays an important role in the reasoning process. Since language models are prone to various types of errors during the reasoning process, PRMs are required to possess nuanced capabilities for detecting various implicit error types in real-world scenarios. However, current benchmarks primarily focus on step correctness, failing to evaluate PRMs' performance systematically. To address this gap, we introduce PRMBench, a process-level benchmark specifically designed to asse"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.03124","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.03124/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.03124","created_at":"2026-07-05T11:28:33.281535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.03124v5","created_at":"2026-07-05T11:28:33.281535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.03124","created_at":"2026-07-05T11:28:33.281535+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZGXPLEYEGD7R","created_at":"2026-07-05T11:28:33.281535+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZGXPLEYEGD7RFJFD","created_at":"2026-07-05T11:28:33.281535+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZGXPLEYE","created_at":"2026-07-05T11:28:33.281535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09078","citing_title":"The Hidden Bias of Process Reward Models:PRISM for Rewarding the Right Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07367","citing_title":"Self-evolving LLM agents with in-distribution Optimization","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":173,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12384","citing_title":"Scalable Token-Level Hallucination Detection in Large Language Models","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV","json":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV.json","graph_json":"https://pith.science/api/pith-number/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/graph.json","events_json":"https://pith.science/api/pith-number/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/events.json","paper":"https://pith.science/paper/ZGXPLEYE"},"agent_actions":{"view_html":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV","download_json":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV.json","view_paper":"https://pith.science/paper/ZGXPLEYE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.03124&json=true","fetch_graph":"https://pith.science/api/pith-number/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/graph.json","fetch_events":"https://pith.science/api/pith-number/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/action/storage_attestation","attest_author":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/action/author_attestation","sign_citation":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/action/citation_signature","submit_replication":"https://pith.science/pith/ZGXPLEYEGD7RFJFDZ2UFCCRAMV/action/replication_record"}},"created_at":"2026-07-05T11:28:33.281535+00:00","updated_at":"2026-07-05T11:28:33.281535+00:00"}