{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:73Z5GYZQ72JWUYEZJC4O3252AM","short_pith_number":"pith:73Z5GYZQ","schema_version":"1.0","canonical_sha256":"fef3d36330fe936a609948b8edebba030a323a4837a3451b84b96de539499006","source":{"kind":"arxiv","id":"2402.17644","version":2},"attestation_state":"computed","paper":{"title":"Are LLMs Capable of Data-based Statistical and Causal Reasoning? Benchmarking Advanced Quantitative Reasoning with Data","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Kai-Wei Chang, Pan Lu, Xiao Liu, Xueqing Wu, Yansong Feng, Zirui Wu","submitted_at":"2024-02-27T16:15:03Z","abstract_excerpt":"Quantitative reasoning is a critical skill to analyze data, yet the assessment of such ability remains limited. To address this gap, we introduce the Quantitative Reasoning with Data (QRData) benchmark, aiming to evaluate Large Language Models' capability in statistical and causal reasoning with real-world data. The benchmark comprises a carefully constructed dataset of 411 questions accompanied by data sheets from textbooks, online learning materials, and academic papers. To compare models' quantitative reasoning abilities on data and text, we enrich the benchmark with an auxiliary set of 290"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17644","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-27T16:15:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"292b092a7d33625190211e98194cf1004178aa1d07db19e5d28b3d6d44eb4db5","abstract_canon_sha256":"3720e4cb599a1afbcf48608ffcf1cd36373afd1d230d95ea117627c7afde04fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:23.096299Z","signature_b64":"3FsPFAbGcFC9Deovsu1VdZjSttkQYh+Sl47Jcruax5o0hDmDiCdWK+PXyvNjMVCIdixdyQ7LTm8E2Dr8xY2TAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fef3d36330fe936a609948b8edebba030a323a4837a3451b84b96de539499006","last_reissued_at":"2026-07-05T08:29:23.095860Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:23.095860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are LLMs Capable of Data-based Statistical and Causal Reasoning? Benchmarking Advanced Quantitative Reasoning with Data","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Kai-Wei Chang, Pan Lu, Xiao Liu, Xueqing Wu, Yansong Feng, Zirui Wu","submitted_at":"2024-02-27T16:15:03Z","abstract_excerpt":"Quantitative reasoning is a critical skill to analyze data, yet the assessment of such ability remains limited. To address this gap, we introduce the Quantitative Reasoning with Data (QRData) benchmark, aiming to evaluate Large Language Models' capability in statistical and causal reasoning with real-world data. The benchmark comprises a carefully constructed dataset of 411 questions accompanied by data sheets from textbooks, online learning materials, and academic papers. To compare models' quantitative reasoning abilities on data and text, we enrich the benchmark with an auxiliary set of 290"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17644","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17644","created_at":"2026-07-05T08:29:23.095919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17644v2","created_at":"2026-07-05T08:29:23.095919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17644","created_at":"2026-07-05T08:29:23.095919+00:00"},{"alias_kind":"pith_short_12","alias_value":"73Z5GYZQ72JW","created_at":"2026-07-05T08:29:23.095919+00:00"},{"alias_kind":"pith_short_16","alias_value":"73Z5GYZQ72JWUYEZ","created_at":"2026-07-05T08:29:23.095919+00:00"},{"alias_kind":"pith_short_8","alias_value":"73Z5GYZQ","created_at":"2026-07-05T08:29:23.095919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.04509","citing_title":"ErrorRadar: Benchmarking Complex Mathematical Reasoning of Multimodal Large Language Models Via Error Detection","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09741","citing_title":"FoNE: Precise Single-Token Number Embeddings via Fourier Features","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11348","citing_title":"Large Language Models for Causal Relations Extraction in Social Media: A Validation Framework for Disaster Intelligence","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM","json":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM.json","graph_json":"https://pith.science/api/pith-number/73Z5GYZQ72JWUYEZJC4O3252AM/graph.json","events_json":"https://pith.science/api/pith-number/73Z5GYZQ72JWUYEZJC4O3252AM/events.json","paper":"https://pith.science/paper/73Z5GYZQ"},"agent_actions":{"view_html":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM","download_json":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM.json","view_paper":"https://pith.science/paper/73Z5GYZQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17644&json=true","fetch_graph":"https://pith.science/api/pith-number/73Z5GYZQ72JWUYEZJC4O3252AM/graph.json","fetch_events":"https://pith.science/api/pith-number/73Z5GYZQ72JWUYEZJC4O3252AM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM/action/storage_attestation","attest_author":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM/action/author_attestation","sign_citation":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM/action/citation_signature","submit_replication":"https://pith.science/pith/73Z5GYZQ72JWUYEZJC4O3252AM/action/replication_record"}},"created_at":"2026-07-05T08:29:23.095919+00:00","updated_at":"2026-07-05T08:29:23.095919+00:00"}