{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PWNZGQI46DAA3JDGJGDFBJWFEV","short_pith_number":"pith:PWNZGQI4","schema_version":"1.0","canonical_sha256":"7d9b93411cf0c00da466498650a6c5255688f79c3ccf356a7493801b3582eec8","source":{"kind":"arxiv","id":"2405.00332","version":4},"attestation_state":"computed","paper":{"title":"A Careful Examination of Large Language Model Performance on Grade School Arithmetic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Catherine Wu, Charlotte Zhuang, Dean Lee, Dylan Slack, Hugh Zhang, Jeff Da, Michele Lunati, Pranav Raja, Qin Lyu, Russell Kaplan, Sean Hendryx, Summer Yue, Tiffany Zhao, Vaughn Robinson, Will Song","submitted_at":"2024-05-01T05:52:05Z","abstract_excerpt":"Large language models (LLMs) have achieved impressive success on many benchmarks for mathematical reasoning. However, there is growing concern that some of this performance actually reflects dataset contamination, where data closely resembling benchmark questions leaks into the training data, instead of true reasoning ability. To investigate this claim rigorously, we commission Grade School Math 1000 (GSM1k). GSM1k is designed to mirror the style and complexity of the established GSM8k benchmark, the gold standard for measuring elementary mathematical reasoning. We ensure that the two benchmar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.00332","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-01T05:52:05Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"99aa56fec77c1c0e6ca55815bf7eae7d2269ef7d222ec0b07fb9ced9d31ac0d4","abstract_canon_sha256":"bf05f2b667fd90f7735da877121a5ea617547691104587b6bc99d992d900c258"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:22.852673Z","signature_b64":"hn4j2bXVVuF3WpHORgXdKmBfdM80BsBFtuoVBG/QTu2IcH15Ueq+yoNnhy469fc9pYBN5rgN4TETrf73n80XCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d9b93411cf0c00da466498650a6c5255688f79c3ccf356a7493801b3582eec8","last_reissued_at":"2026-07-05T09:39:22.852176Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:22.852176Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Careful Examination of Large Language Model Performance on Grade School Arithmetic","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Catherine Wu, Charlotte Zhuang, Dean Lee, Dylan Slack, Hugh Zhang, Jeff Da, Michele Lunati, Pranav Raja, Qin Lyu, Russell Kaplan, Sean Hendryx, Summer Yue, Tiffany Zhao, Vaughn Robinson, Will Song","submitted_at":"2024-05-01T05:52:05Z","abstract_excerpt":"Large language models (LLMs) have achieved impressive success on many benchmarks for mathematical reasoning. However, there is growing concern that some of this performance actually reflects dataset contamination, where data closely resembling benchmark questions leaks into the training data, instead of true reasoning ability. To investigate this claim rigorously, we commission Grade School Math 1000 (GSM1k). GSM1k is designed to mirror the style and complexity of the established GSM8k benchmark, the gold standard for measuring elementary mathematical reasoning. We ensure that the two benchmar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00332","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.00332/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.00332","created_at":"2026-07-05T09:39:22.852232+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.00332v4","created_at":"2026-07-05T09:39:22.852232+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00332","created_at":"2026-07-05T09:39:22.852232+00:00"},{"alias_kind":"pith_short_12","alias_value":"PWNZGQI46DAA","created_at":"2026-07-05T09:39:22.852232+00:00"},{"alias_kind":"pith_short_16","alias_value":"PWNZGQI46DAA3JDG","created_at":"2026-07-05T09:39:22.852232+00:00"},{"alias_kind":"pith_short_8","alias_value":"PWNZGQI4","created_at":"2026-07-05T09:39:22.852232+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22678","citing_title":"RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00276","citing_title":"Testing Frontier Large Language Models' Physics Literacy in Parallel Physical Worlds","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22678","citing_title":"RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2406.19314","citing_title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05229","citing_title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16392","citing_title":"RoMathExam: A Longitudinal Dataset of Romanian Math Exams (1895-2025) with a Seven-Decade Core (1957-2025)","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16941","citing_title":"SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks?","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05561","citing_title":"BitCal-TTS: Bit-Calibrated Test-Time Scaling for Quantized Reasoning Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06865","citing_title":"Dataset Watermarking for Closed LLMs with Provable Detection","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV","json":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV.json","graph_json":"https://pith.science/api/pith-number/PWNZGQI46DAA3JDGJGDFBJWFEV/graph.json","events_json":"https://pith.science/api/pith-number/PWNZGQI46DAA3JDGJGDFBJWFEV/events.json","paper":"https://pith.science/paper/PWNZGQI4"},"agent_actions":{"view_html":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV","download_json":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV.json","view_paper":"https://pith.science/paper/PWNZGQI4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.00332&json=true","fetch_graph":"https://pith.science/api/pith-number/PWNZGQI46DAA3JDGJGDFBJWFEV/graph.json","fetch_events":"https://pith.science/api/pith-number/PWNZGQI46DAA3JDGJGDFBJWFEV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV/action/storage_attestation","attest_author":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV/action/author_attestation","sign_citation":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV/action/citation_signature","submit_replication":"https://pith.science/pith/PWNZGQI46DAA3JDGJGDFBJWFEV/action/replication_record"}},"created_at":"2026-07-05T09:39:22.852232+00:00","updated_at":"2026-07-05T09:39:22.852232+00:00"}