{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:QX54DYPCC6K2SOE5PGA2FLKF5W","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6b55af567158883ee42906c7bba33746c408fdc498d173889a9ba8cc125bca4c","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-06-02T16:32:53Z","title_canon_sha256":"946c061d964999e9756cf798bc2406be05cf7fd63d7b284aa25749d1fd56906d"},"schema_version":"1.0","source":{"id":"2606.03858","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2606.03858","created_at":"2026-06-03T02:06:04Z"},{"alias_kind":"arxiv_version","alias_value":"2606.03858v1","created_at":"2026-06-03T02:06:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.03858","created_at":"2026-06-03T02:06:04Z"},{"alias_kind":"pith_short_12","alias_value":"QX54DYPCC6K2","created_at":"2026-06-03T02:06:04Z"},{"alias_kind":"pith_short_16","alias_value":"QX54DYPCC6K2SOE5","created_at":"2026-06-03T02:06:04Z"},{"alias_kind":"pith_short_8","alias_value":"QX54DYPC","created_at":"2026-06-03T02:06:04Z"}],"graph_snapshots":[{"event_id":"sha256:7da1d969e50e11848f479397fb5c4278f8412266b726547d043684ccab64207e","target":"graph","created_at":"2026-06-03T02:06:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2606.03858/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Despite the pivotal role of numerical reasoning as the cornerstone of mathematical capabilities in large language models (LLMs) across applications, few benchmarks evaluate LLMs by integrating numerical processing and mathematical reasoning, hindering the interpretability of failures in math tasks. We introduce PyraMathBench, a comprehensive hierarchical benchmark with 32,505 questions derived from 7,404 math word problems, spanning 4 key cognitive aspects, 14 subcategories, and 2 modalities. Experiments reveal that LLMs' performance is severely compromised by inadequate numerical computation ","authors_text":"Gerard de Melo, Liang He, Linlin Wang, Zetian Ouyang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-06-02T16:32:53Z","title":"PyraMathBench: Evaluating and Improving Mathematical Capability in Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.03858","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4bd42858ebcc8f4cc8741c8814401012ba5d50911b863791dac2abd4c356d224","target":"record","created_at":"2026-06-03T02:06:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6b55af567158883ee42906c7bba33746c408fdc498d173889a9ba8cc125bca4c","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-06-02T16:32:53Z","title_canon_sha256":"946c061d964999e9756cf798bc2406be05cf7fd63d7b284aa25749d1fd56906d"},"schema_version":"1.0","source":{"id":"2606.03858","kind":"arxiv","version":1}},"canonical_sha256":"85fbc1e1e21795a9389d7981a2ad45ed90acf4868b61007b1a3d8131021e39f6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"85fbc1e1e21795a9389d7981a2ad45ed90acf4868b61007b1a3d8131021e39f6","first_computed_at":"2026-06-03T02:06:04.447011Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-03T02:06:04.447011Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ECqy0GN4GSJpl6OKBhYOGlyuQPdtZJ2SSI/fT5gaFQXwgzvtSISQXI1R1UtcC+UyyqrkaENdaxxeRHMsttD2Bw==","signature_status":"signed_v1","signed_at":"2026-06-03T02:06:04.447423Z","signed_message":"canonical_sha256_bytes"},"source_id":"2606.03858","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4bd42858ebcc8f4cc8741c8814401012ba5d50911b863791dac2abd4c356d224","sha256:7da1d969e50e11848f479397fb5c4278f8412266b726547d043684ccab64207e"],"state_sha256":"078e74a241a8aa64068cb379e40fb65d474407d079d20f09f95f2a94a7063337"}