{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DYYRLCTBEN27F67AQUWLLIWY5G","short_pith_number":"pith:DYYRLCTB","schema_version":"1.0","canonical_sha256":"1e31158a612375f2fbe0852cb5a2d8e9b19a792bcc9035f4c7aa8653047708ff","source":{"kind":"arxiv","id":"2412.06559","version":4},"attestation_state":"computed","paper":{"title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Beichen Zhang, Bowen Yu, Chujie Zheng, Dayiheng Liu, Jingren Zhou, Junyang Lin, Keming Lu, Runji Lin, Zhenru Zhang","submitted_at":"2024-12-09T15:11:40Z","abstract_excerpt":"As language models regularly make mistakes when solving math problems, automated identification of errors in the reasoning process becomes increasingly significant for their scalable oversight. In this paper, we introduce ProcessBench for measuring the ability to identify erroneous steps in mathematical reasoning. It consists of 3,400 test cases, primarily focused on competition- and Olympiad-level math problems. Each test case contains a step-by-step solution with error location annotated by human experts. Models are required to identify the earliest step that contains an error, or conclude t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06559","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-12-09T15:11:40Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"fefd676b10464cc12f97b8601f0539fd6ad9f82c92c5856b575102250bef41b7","abstract_canon_sha256":"45f46ca5b11b6c2c7854efd428af7d7a5c9d59d3fae93e3c017831769d2a4d55"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:44.355785Z","signature_b64":"sE+PQpso+Pl8FRw8yY4AVZraOE0IlAzChN/q1I0ZQrz6lqDc7CzYEwKh/IUwMLH4TY1+0ma4qNg4vDSBQodzAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e31158a612375f2fbe0852cb5a2d8e9b19a792bcc9035f4c7aa8653047708ff","last_reissued_at":"2026-07-05T11:09:44.355284Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:44.355284Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Beichen Zhang, Bowen Yu, Chujie Zheng, Dayiheng Liu, Jingren Zhou, Junyang Lin, Keming Lu, Runji Lin, Zhenru Zhang","submitted_at":"2024-12-09T15:11:40Z","abstract_excerpt":"As language models regularly make mistakes when solving math problems, automated identification of errors in the reasoning process becomes increasingly significant for their scalable oversight. In this paper, we introduce ProcessBench for measuring the ability to identify erroneous steps in mathematical reasoning. It consists of 3,400 test cases, primarily focused on competition- and Olympiad-level math problems. Each test case contains a step-by-step solution with error location annotated by human experts. Models are required to identify the earliest step that contains an error, or conclude t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06559","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06559/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06559","created_at":"2026-07-05T11:09:44.355351+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06559v4","created_at":"2026-07-05T11:09:44.355351+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06559","created_at":"2026-07-05T11:09:44.355351+00:00"},{"alias_kind":"pith_short_12","alias_value":"DYYRLCTBEN27","created_at":"2026-07-05T11:09:44.355351+00:00"},{"alias_kind":"pith_short_16","alias_value":"DYYRLCTBEN27F67A","created_at":"2026-07-05T11:09:44.355351+00:00"},{"alias_kind":"pith_short_8","alias_value":"DYYRLCTB","created_at":"2026-07-05T11:09:44.355351+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22317","citing_title":"Curriculum Reinforcement Learning Can Incentivize Reasoning Capacity in LLMs Beyond the Base Model","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30851","citing_title":"Test-Time Verification for Text-to-SQL via Outcome Reward Models","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27712","citing_title":"Prefix-Safe Bayesian Belief Tracking for LLM Reasoning Reliability:Separating Calibration from Ranking","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":261,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2508.03556","citing_title":"VRPRM: Process Reward Modeling via Visual Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22102","citing_title":"ExComm: Exploration-Stage Communication for Error-Resilient Agentic Test-Time Scaling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15698","citing_title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03403","citing_title":"Beyond Correctness: Harmonizing Process and Outcome Rewards through RL Training","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14004","citing_title":"Early Stopping Chain-of-thoughts in Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19228","citing_title":"Diagnosing Multi-step Reasoning Failures in Black-box LLMs via Stepwise Confidence Attribution","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01203","citing_title":"Attention Sink Forges Native MoE in Attention Layers: Sink-Aware Training to Address Head Collapse","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03332","citing_title":"Fragile Thoughts: How Large Language Models Handle Chain-of-Thought Perturbations","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19678","citing_title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13772","citing_title":"Where Does Reasoning Break? Step-Level Hallucination Detection via Hidden-State Transport Geometry","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05366","citing_title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21611","citing_title":"Process Supervision via Verbal Critique Improves Reasoning in Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21510","citing_title":"OptiVerse: A Comprehensive Benchmark towards Optimization Problem Solving","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20183","citing_title":"Dual-Cluster Memory Agent: Resolving Multi-Paradigm Ambiguity in Optimization Problem Solving","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18464","citing_title":"Semantic Step Prediction: Multi-Step Latent Forecasting in LLM Reasoning Trajectories via Step Sampling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17282","citing_title":"MedPRMBench: A Fine-grained Benchmark for Process Reward Models in Medical Reasoning","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G","json":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G.json","graph_json":"https://pith.science/api/pith-number/DYYRLCTBEN27F67AQUWLLIWY5G/graph.json","events_json":"https://pith.science/api/pith-number/DYYRLCTBEN27F67AQUWLLIWY5G/events.json","paper":"https://pith.science/paper/DYYRLCTB"},"agent_actions":{"view_html":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G","download_json":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G.json","view_paper":"https://pith.science/paper/DYYRLCTB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06559&json=true","fetch_graph":"https://pith.science/api/pith-number/DYYRLCTBEN27F67AQUWLLIWY5G/graph.json","fetch_events":"https://pith.science/api/pith-number/DYYRLCTBEN27F67AQUWLLIWY5G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G/action/storage_attestation","attest_author":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G/action/author_attestation","sign_citation":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G/action/citation_signature","submit_replication":"https://pith.science/pith/DYYRLCTBEN27F67AQUWLLIWY5G/action/replication_record"}},"created_at":"2026-07-05T11:09:44.355351+00:00","updated_at":"2026-07-05T11:09:44.355351+00:00"}