{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QXDH4M6WEUAHZBAZDMGUETQIRW","short_pith_number":"pith:QXDH4M6W","schema_version":"1.0","canonical_sha256":"85c67e33d625007c84191b0d424e088da9ca8b7749aae9159f23aaf124f22252","source":{"kind":"arxiv","id":"2404.05692","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Mathematical Reasoning Beyond Accuracy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Pengfei Liu, Shijie Xia, Tongshuang Wu, Xuefeng Li, Yixin Liu","submitted_at":"2024-04-08T17:18:04Z","abstract_excerpt":"The leaderboard of Large Language Models (LLMs) in mathematical tasks has been continuously updated. However, the majority of evaluations focus solely on the final results, neglecting the quality of the intermediate steps. This oversight can mask underlying problems, such as logical errors or unnecessary steps in the reasoning process. To measure reasoning beyond final-answer accuracy, we introduce ReasonEval, a new methodology for evaluating the quality of reasoning steps. ReasonEval employs validity and redundancy to characterize the reasoning quality, as well as accompanying LLMs to assess "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05692","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-08T17:18:04Z","cross_cats_sorted":[],"title_canon_sha256":"555267f58550ff6d46dbfc68b3dc01ac3bf714c53e41b8f7e354b389f13908f5","abstract_canon_sha256":"d8e6b3ada393c45cc796a74c4a2cde6b3f3a8b9bb2c050bf7c2a7c297c6d11fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:35.857176Z","signature_b64":"NF0YBB1Ci8ocA4//gMop+J7+slIp6kRww1/5+/NaWxllRs1a3OpDj7zPQfYRhEJGs6Ngmml99HnQGcntZ+SKCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85c67e33d625007c84191b0d424e088da9ca8b7749aae9159f23aaf124f22252","last_reissued_at":"2026-07-05T10:00:35.856703Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:35.856703Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Mathematical Reasoning Beyond Accuracy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Pengfei Liu, Shijie Xia, Tongshuang Wu, Xuefeng Li, Yixin Liu","submitted_at":"2024-04-08T17:18:04Z","abstract_excerpt":"The leaderboard of Large Language Models (LLMs) in mathematical tasks has been continuously updated. However, the majority of evaluations focus solely on the final results, neglecting the quality of the intermediate steps. This oversight can mask underlying problems, such as logical errors or unnecessary steps in the reasoning process. To measure reasoning beyond final-answer accuracy, we introduce ReasonEval, a new methodology for evaluating the quality of reasoning steps. ReasonEval employs validity and redundancy to characterize the reasoning quality, as well as accompanying LLMs to assess "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05692","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05692/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05692","created_at":"2026-07-05T10:00:35.856767+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05692v2","created_at":"2026-07-05T10:00:35.856767+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05692","created_at":"2026-07-05T10:00:35.856767+00:00"},{"alias_kind":"pith_short_12","alias_value":"QXDH4M6WEUAH","created_at":"2026-07-05T10:00:35.856767+00:00"},{"alias_kind":"pith_short_16","alias_value":"QXDH4M6WEUAHZBAZ","created_at":"2026-07-05T10:00:35.856767+00:00"},{"alias_kind":"pith_short_8","alias_value":"QXDH4M6W","created_at":"2026-07-05T10:00:35.856767+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.02079","citing_title":"Argumentative Large Language Models for Explainable and Contestable Claim Verification","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04588","citing_title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15698","citing_title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04588","citing_title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2501.07301","citing_title":"The Lessons of Developing Process Reward Models in Mathematical Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05366","citing_title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":255,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17282","citing_title":"MedPRMBench: A Fine-grained Benchmark for Process Reward Models in Medical Reasoning","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW","json":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW.json","graph_json":"https://pith.science/api/pith-number/QXDH4M6WEUAHZBAZDMGUETQIRW/graph.json","events_json":"https://pith.science/api/pith-number/QXDH4M6WEUAHZBAZDMGUETQIRW/events.json","paper":"https://pith.science/paper/QXDH4M6W"},"agent_actions":{"view_html":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW","download_json":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW.json","view_paper":"https://pith.science/paper/QXDH4M6W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05692&json=true","fetch_graph":"https://pith.science/api/pith-number/QXDH4M6WEUAHZBAZDMGUETQIRW/graph.json","fetch_events":"https://pith.science/api/pith-number/QXDH4M6WEUAHZBAZDMGUETQIRW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW/action/storage_attestation","attest_author":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW/action/author_attestation","sign_citation":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW/action/citation_signature","submit_replication":"https://pith.science/pith/QXDH4M6WEUAHZBAZDMGUETQIRW/action/replication_record"}},"created_at":"2026-07-05T10:00:35.856767+00:00","updated_at":"2026-07-05T10:00:35.856767+00:00"}