{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XCQG3UBVSLWGY3NY462FIV4IEO","short_pith_number":"pith:XCQG3UBV","schema_version":"1.0","canonical_sha256":"b8a06dd03592ec6c6db8e7b45457882399d0227b84c7548cbe7fc4ed29d74074","source":{"kind":"arxiv","id":"2311.02216","version":1},"attestation_state":"computed","paper":{"title":"Exploring the Numerical Reasoning Capabilities of Language Models: A Comprehensive Analysis on Tabular Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhilash Shankarampeta, Arpit Patil, Elena Simperl, Mubashara Akhtar, Oana Cocarascu, Vivek Gupta","submitted_at":"2023-11-03T20:05:30Z","abstract_excerpt":"Numbers are crucial for various real-world domains such as finance, economics, and science. Thus, understanding and reasoning with numbers are essential skills for language models to solve different tasks. While different numerical benchmarks have been introduced in recent years, they are limited to specific numerical aspects mostly. In this paper, we propose a hierarchical taxonomy for numerical reasoning skills with more than ten reasoning types across four levels: representation, number sense, manipulation, and complex reasoning. We conduct a comprehensive evaluation of state-of-the-art mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.02216","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-03T20:05:30Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"4f11c8ba6ea2f2b4df2bf5185cb08f74ea16cd47fbd6f56f7551f96a17f97346","abstract_canon_sha256":"38e2779457834e29287e412e5fb88333643b04ad0d6ccd85e3eed437c415d3c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:07.637512Z","signature_b64":"dn11BXLCzHbDgszctpBt5CHh8KdLmY0nDAaRhFlWbl8nR95V4mY2Sbe2XdfUZ6eVqOgptxmBnzhgfL3uNQ0NDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8a06dd03592ec6c6db8e7b45457882399d0227b84c7548cbe7fc4ed29d74074","last_reissued_at":"2026-07-05T07:09:07.637009Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:07.637009Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the Numerical Reasoning Capabilities of Language Models: A Comprehensive Analysis on Tabular Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhilash Shankarampeta, Arpit Patil, Elena Simperl, Mubashara Akhtar, Oana Cocarascu, Vivek Gupta","submitted_at":"2023-11-03T20:05:30Z","abstract_excerpt":"Numbers are crucial for various real-world domains such as finance, economics, and science. Thus, understanding and reasoning with numbers are essential skills for language models to solve different tasks. While different numerical benchmarks have been introduced in recent years, they are limited to specific numerical aspects mostly. In this paper, we propose a hierarchical taxonomy for numerical reasoning skills with more than ten reasoning types across four levels: representation, number sense, manipulation, and complex reasoning. We conduct a comprehensive evaluation of state-of-the-art mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02216","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02216/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.02216","created_at":"2026-07-05T07:09:07.637070+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.02216v1","created_at":"2026-07-05T07:09:07.637070+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02216","created_at":"2026-07-05T07:09:07.637070+00:00"},{"alias_kind":"pith_short_12","alias_value":"XCQG3UBVSLWG","created_at":"2026-07-05T07:09:07.637070+00:00"},{"alias_kind":"pith_short_16","alias_value":"XCQG3UBVSLWGY3NY","created_at":"2026-07-05T07:09:07.637070+00:00"},{"alias_kind":"pith_short_8","alias_value":"XCQG3UBV","created_at":"2026-07-05T07:09:07.637070+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29799","citing_title":"The CRISTAL Method: Neurosymbolic analysis from AI-synthesized world models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01101","citing_title":"TSVer: A Benchmark for Fact Verification Against Time-Series Evidence","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO","json":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO.json","graph_json":"https://pith.science/api/pith-number/XCQG3UBVSLWGY3NY462FIV4IEO/graph.json","events_json":"https://pith.science/api/pith-number/XCQG3UBVSLWGY3NY462FIV4IEO/events.json","paper":"https://pith.science/paper/XCQG3UBV"},"agent_actions":{"view_html":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO","download_json":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO.json","view_paper":"https://pith.science/paper/XCQG3UBV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.02216&json=true","fetch_graph":"https://pith.science/api/pith-number/XCQG3UBVSLWGY3NY462FIV4IEO/graph.json","fetch_events":"https://pith.science/api/pith-number/XCQG3UBVSLWGY3NY462FIV4IEO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO/action/storage_attestation","attest_author":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO/action/author_attestation","sign_citation":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO/action/citation_signature","submit_replication":"https://pith.science/pith/XCQG3UBVSLWGY3NY462FIV4IEO/action/replication_record"}},"created_at":"2026-07-05T07:09:07.637070+00:00","updated_at":"2026-07-05T07:09:07.637070+00:00"}