{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5NGNKDFUPVM3QPWN4RXAWWLGPE","short_pith_number":"pith:5NGNKDFU","schema_version":"1.0","canonical_sha256":"eb4cd50cb47d59b83ecde46e0b59667928ec7e769ffa324b463e20a8fcfe6375","source":{"kind":"arxiv","id":"2410.03117","version":1},"attestation_state":"computed","paper":{"title":"ProcBench: Benchmark for Multi-Step Reasoning and Following Procedure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Hiroki Ikoma, Hiroki Seto, Ippei Fujisawa, Pei-Chun Chien, Rina Onda, Ryota Kanai, Sensho Nobe, Yoshiaki Uchida","submitted_at":"2024-10-04T03:21:24Z","abstract_excerpt":"Reasoning is central to a wide range of intellectual activities, and while the capabilities of large language models (LLMs) continue to advance, their performance in reasoning tasks remains limited. The processes and mechanisms underlying reasoning are not yet fully understood, but key elements include path exploration, selection of relevant knowledge, and multi-step inference. Problems are solved through the synthesis of these components. In this paper, we propose a benchmark that focuses on a specific aspect of reasoning ability: the direct evaluation of multi-step inference. To this end, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03117","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-04T03:21:24Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"e61e91e0d31d51819ae099585367f38ae3dd50f939f8b6a8a74ebb9dcbcb97cd","abstract_canon_sha256":"69d1a679ede03e6038c2e1e3e258b4a9218890e1d347f65b3372b84cf8902d22"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:44.726909Z","signature_b64":"4ya0alxU98Gm+Cyr3NuXdRgRX9Sz2LNi7Ob28BLV2BC9BsNSxn4aU8bmaf4eWfqALncIRNj62SFDLcnR488bBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb4cd50cb47d59b83ecde46e0b59667928ec7e769ffa324b463e20a8fcfe6375","last_reissued_at":"2026-07-05T09:15:44.726409Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:44.726409Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ProcBench: Benchmark for Multi-Step Reasoning and Following Procedure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Hiroki Ikoma, Hiroki Seto, Ippei Fujisawa, Pei-Chun Chien, Rina Onda, Ryota Kanai, Sensho Nobe, Yoshiaki Uchida","submitted_at":"2024-10-04T03:21:24Z","abstract_excerpt":"Reasoning is central to a wide range of intellectual activities, and while the capabilities of large language models (LLMs) continue to advance, their performance in reasoning tasks remains limited. The processes and mechanisms underlying reasoning are not yet fully understood, but key elements include path exploration, selection of relevant knowledge, and multi-step inference. Problems are solved through the synthesis of these components. In this paper, we propose a benchmark that focuses on a specific aspect of reasoning ability: the direct evaluation of multi-step inference. To this end, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03117","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03117/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03117","created_at":"2026-07-05T09:15:44.726473+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03117v1","created_at":"2026-07-05T09:15:44.726473+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03117","created_at":"2026-07-05T09:15:44.726473+00:00"},{"alias_kind":"pith_short_12","alias_value":"5NGNKDFUPVM3","created_at":"2026-07-05T09:15:44.726473+00:00"},{"alias_kind":"pith_short_16","alias_value":"5NGNKDFUPVM3QPWN","created_at":"2026-07-05T09:15:44.726473+00:00"},{"alias_kind":"pith_short_8","alias_value":"5NGNKDFU","created_at":"2026-07-05T09:15:44.726473+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12767","citing_title":"Constructing Evaluation Datasets for Procedural Reasoning: Balancing Naturalness, Grounding, and Multi-Hop Coverage","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10722","citing_title":"Continual LLM Upcycling: A Predictor-Gated Bank-Wise Sparsity Training Recipe for Dense-to-Sparse LLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20251","citing_title":"ProcCtrlBench: Evaluating Process-Level Defects and Control Preservation in LLM Coding Agents","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE","json":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE.json","graph_json":"https://pith.science/api/pith-number/5NGNKDFUPVM3QPWN4RXAWWLGPE/graph.json","events_json":"https://pith.science/api/pith-number/5NGNKDFUPVM3QPWN4RXAWWLGPE/events.json","paper":"https://pith.science/paper/5NGNKDFU"},"agent_actions":{"view_html":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE","download_json":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE.json","view_paper":"https://pith.science/paper/5NGNKDFU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03117&json=true","fetch_graph":"https://pith.science/api/pith-number/5NGNKDFUPVM3QPWN4RXAWWLGPE/graph.json","fetch_events":"https://pith.science/api/pith-number/5NGNKDFUPVM3QPWN4RXAWWLGPE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE/action/storage_attestation","attest_author":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE/action/author_attestation","sign_citation":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE/action/citation_signature","submit_replication":"https://pith.science/pith/5NGNKDFUPVM3QPWN4RXAWWLGPE/action/replication_record"}},"created_at":"2026-07-05T09:15:44.726473+00:00","updated_at":"2026-07-05T09:15:44.726473+00:00"}