{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HEPX2RTBAKR5XLWA2SJKSAITIH","short_pith_number":"pith:HEPX2RTB","schema_version":"1.0","canonical_sha256":"391f7d466102a3dbaec0d492a9011341d3d6661c5f25a578f41f92ee0a1a8382","source":{"kind":"arxiv","id":"2307.03381","version":1},"attestation_state":"computed","paper":{"title":"Teaching Arithmetic to Small Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dimitris Papailiopoulos, Jason D. Lee, Kangwook Lee, Kartik Sreenivasan, Nayoung Lee","submitted_at":"2023-07-07T04:33:31Z","abstract_excerpt":"Large language models like GPT-4 exhibit emergent capabilities across general-purpose tasks, such as basic arithmetic, when trained on extensive text data, even though these tasks are not explicitly encoded by the unsupervised, next-token prediction objective. This study investigates how small transformers, trained from random initialization, can efficiently learn arithmetic operations such as addition, multiplication, and elementary functions like square root, using the next-token prediction objective. We first demonstrate that conventional training data is not the most effective for arithmet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.03381","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-07T04:33:31Z","cross_cats_sorted":[],"title_canon_sha256":"88cf897613daba6cb5a94640aa5bdaadfb938ad6286bcea97f369273a66715ac","abstract_canon_sha256":"157f0e3fddd0e8980f8087741878952b42ee7ce435dbc1a2e368ce7c502bbdcb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:28:42.106128Z","signature_b64":"Mlc51v+Z2aH37G03JCSE6LoGCRLRxvU3LEpbCULcRFxBE2Ouw4BBcdX+x/cj5bJc2FgcFKj3J7gEtatGt4VkBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"391f7d466102a3dbaec0d492a9011341d3d6661c5f25a578f41f92ee0a1a8382","last_reissued_at":"2026-07-05T06:28:42.105712Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:28:42.105712Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Teaching Arithmetic to Small Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dimitris Papailiopoulos, Jason D. Lee, Kangwook Lee, Kartik Sreenivasan, Nayoung Lee","submitted_at":"2023-07-07T04:33:31Z","abstract_excerpt":"Large language models like GPT-4 exhibit emergent capabilities across general-purpose tasks, such as basic arithmetic, when trained on extensive text data, even though these tasks are not explicitly encoded by the unsupervised, next-token prediction objective. This study investigates how small transformers, trained from random initialization, can efficiently learn arithmetic operations such as addition, multiplication, and elementary functions like square root, using the next-token prediction objective. We first demonstrate that conventional training data is not the most effective for arithmet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.03381","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.03381/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.03381","created_at":"2026-07-05T06:28:42.105778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.03381v1","created_at":"2026-07-05T06:28:42.105778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.03381","created_at":"2026-07-05T06:28:42.105778+00:00"},{"alias_kind":"pith_short_12","alias_value":"HEPX2RTBAKR5","created_at":"2026-07-05T06:28:42.105778+00:00"},{"alias_kind":"pith_short_16","alias_value":"HEPX2RTBAKR5XLWA","created_at":"2026-07-05T06:28:42.105778+00:00"},{"alias_kind":"pith_short_8","alias_value":"HEPX2RTB","created_at":"2026-07-05T06:28:42.105778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05106","citing_title":"Arithmetic Pedagogy for Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08022","citing_title":"Globally Optimal Training of Spiking Neural Networks via Parameter Reconstruction","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14659","citing_title":"Slower Generalization, Faster Memorization: A Sweet Spot in Algorithmic Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09741","citing_title":"FoNE: Precise Single-Token Number Embeddings via Fourier Features","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16675","citing_title":"LinAlg-Bench: A Forensic Benchmark Revealing Structural Failure Modes in LLM Mathematical Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03682","citing_title":"From Implicit to Explicit: Token-Efficient Logical Supervision for Mathematical Reasoning in LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08022","citing_title":"Globally Optimal Training of Spiking Neural Networks via Parameter Reconstruction","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":93,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH","json":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH.json","graph_json":"https://pith.science/api/pith-number/HEPX2RTBAKR5XLWA2SJKSAITIH/graph.json","events_json":"https://pith.science/api/pith-number/HEPX2RTBAKR5XLWA2SJKSAITIH/events.json","paper":"https://pith.science/paper/HEPX2RTB"},"agent_actions":{"view_html":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH","download_json":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH.json","view_paper":"https://pith.science/paper/HEPX2RTB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.03381&json=true","fetch_graph":"https://pith.science/api/pith-number/HEPX2RTBAKR5XLWA2SJKSAITIH/graph.json","fetch_events":"https://pith.science/api/pith-number/HEPX2RTBAKR5XLWA2SJKSAITIH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH/action/storage_attestation","attest_author":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH/action/author_attestation","sign_citation":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH/action/citation_signature","submit_replication":"https://pith.science/pith/HEPX2RTBAKR5XLWA2SJKSAITIH/action/replication_record"}},"created_at":"2026-07-05T06:28:42.105778+00:00","updated_at":"2026-07-05T06:28:42.105778+00:00"}