{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LH64GGGIP7TKWELJMXSYLZROUM","short_pith_number":"pith:LH64GGGI","schema_version":"1.0","canonical_sha256":"59fdc318c87fe6ab116965e585e62ea335fe5d2f7156377361474127b70a3de9","source":{"kind":"arxiv","id":"2402.14903","version":1},"attestation_state":"computed","paper":{"title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaditya K. Singh, DJ Strouse","submitted_at":"2024-02-22T18:14:09Z","abstract_excerpt":"Tokenization, the division of input text into input tokens, is an often overlooked aspect of the large language model (LLM) pipeline and could be the source of useful or harmful inductive biases. Historically, LLMs have relied on byte pair encoding, without care to specific input domains. With the increased use of LLMs for reasoning, various number-specific tokenization schemes have been adopted, with popular models like LLaMa and PaLM opting for single-digit tokenization while GPT-3.5 and GPT-4 have separate tokens for each 1-, 2-, and 3-digit numbers. In this work, we study the effect this c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14903","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-22T18:14:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a8cf24128907922ef8a835d242ff32d27c57b01ec8f8a8d6c0902eae6b20df72","abstract_canon_sha256":"2df7622c5038b1384babc42d49197c4b71135b6799d042b616369b5e7ee1cfcb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:19.483916Z","signature_b64":"oca7dakx9Xn765a95dEm6o9M/BwaPAmHgGx9UCMNLdEu0JSeXngtGQZ9ZIxFkY4tXriu/NYYwjogfLCObAa8CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59fdc318c87fe6ab116965e585e62ea335fe5d2f7156377361474127b70a3de9","last_reissued_at":"2026-07-05T07:48:19.483493Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:19.483493Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaditya K. Singh, DJ Strouse","submitted_at":"2024-02-22T18:14:09Z","abstract_excerpt":"Tokenization, the division of input text into input tokens, is an often overlooked aspect of the large language model (LLM) pipeline and could be the source of useful or harmful inductive biases. Historically, LLMs have relied on byte pair encoding, without care to specific input domains. With the increased use of LLMs for reasoning, various number-specific tokenization schemes have been adopted, with popular models like LLaMa and PaLM opting for single-digit tokenization while GPT-3.5 and GPT-4 have separate tokens for each 1-, 2-, and 3-digit numbers. In this work, we study the effect this c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14903","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14903/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14903","created_at":"2026-07-05T07:48:19.483548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14903v1","created_at":"2026-07-05T07:48:19.483548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14903","created_at":"2026-07-05T07:48:19.483548+00:00"},{"alias_kind":"pith_short_12","alias_value":"LH64GGGIP7TK","created_at":"2026-07-05T07:48:19.483548+00:00"},{"alias_kind":"pith_short_16","alias_value":"LH64GGGIP7TKWELJ","created_at":"2026-07-05T07:48:19.483548+00:00"},{"alias_kind":"pith_short_8","alias_value":"LH64GGGI","created_at":"2026-07-05T07:48:19.483548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08562","citing_title":"Inside the LLM Word Factory","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28700","citing_title":"The Importance of Being Statistically Earnest: A Critical Re-evaluation of GSM-Symbolic","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03645","citing_title":"The Shape of Addition: Geometric Structures of Arithmetic in Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06824","citing_title":"Efficient numeracy in language models through single-token number embeddings","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12720","citing_title":"FLEXITOKENS: Flexible Tokenization for Evolving Language Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06806","citing_title":"MachineLearningLM: Scaling Many-shot In-context Learning via Continued Pretraining","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12106","citing_title":"Large Language Models as Amortized Pareto-Front Generators for Constrained Bi-Objective Convex Optimization","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11582","citing_title":"A Triadic Suffix Tokenization Scheme for Numerical Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17814","citing_title":"Understanding Secret Leakage Risks in Code LLMs: A Tokenization Perspective","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17105","citing_title":"How Tokenization Limits Phonological Knowledge Representation in Language Models and How to Improve Them","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM","json":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM.json","graph_json":"https://pith.science/api/pith-number/LH64GGGIP7TKWELJMXSYLZROUM/graph.json","events_json":"https://pith.science/api/pith-number/LH64GGGIP7TKWELJMXSYLZROUM/events.json","paper":"https://pith.science/paper/LH64GGGI"},"agent_actions":{"view_html":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM","download_json":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM.json","view_paper":"https://pith.science/paper/LH64GGGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14903&json=true","fetch_graph":"https://pith.science/api/pith-number/LH64GGGIP7TKWELJMXSYLZROUM/graph.json","fetch_events":"https://pith.science/api/pith-number/LH64GGGIP7TKWELJMXSYLZROUM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM/action/storage_attestation","attest_author":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM/action/author_attestation","sign_citation":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM/action/citation_signature","submit_replication":"https://pith.science/pith/LH64GGGIP7TKWELJMXSYLZROUM/action/replication_record"}},"created_at":"2026-07-05T07:48:19.483548+00:00","updated_at":"2026-07-05T07:48:19.483548+00:00"}