{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TO66H2RRGZC3T5BUSORVT77PAY","short_pith_number":"pith:TO66H2RR","schema_version":"1.0","canonical_sha256":"9bbde3ea313645b9f43493a359ffef0637d68fb8ab8da7d198f191bd123585d6","source":{"kind":"arxiv","id":"2512.20757","version":2},"attestation_state":"computed","paper":{"title":"TokSuite: Measuring the Impact of Tokenizer Choice on Language Model Behavior","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Brian Lester, Colin Raffel, Fengyuan Liu, G\\\"ul Sena Alt{\\i}nta\\c{s}, Malikeh Ehghaghi, Marco Ciccone, Wanru Zhao","submitted_at":"2025-12-23T20:43:06Z","abstract_excerpt":"Tokenizers provide the fundamental basis through which text is represented and processed by language models (LMs). Despite the importance of tokenization, its role in LM performance and behavior is poorly understood due to the challenge of measuring the impact of tokenization in isolation. To address this need, we present TokSuite, a collection of models and a benchmark that supports research into tokenization's influence on LMs. Specifically, we release fourteen pre-trained models that use different off-the-shelf tokenizers but are otherwise identical, using the same architecture, dataset, tr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2512.20757","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-12-23T20:43:06Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1e0d068fa47c9e9ec80dfa36ee7ec64f9fc7fdf68f02cb9332d32785d8a25b19","abstract_canon_sha256":"5c5ef1cf3f43bd5562035f1d4c9d10a8210804403e3415b21bbd2599e3b4a45c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:19:44.194959Z","signature_b64":"dvUQKsJcWPwjZn2uvm2mgi0kucbuxQ7Lb7/elcO04dusJCVsumshwv2JOPWgb13dZauhbJRs0ZkThHD3Ryy0DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9bbde3ea313645b9f43493a359ffef0637d68fb8ab8da7d198f191bd123585d6","last_reissued_at":"2026-07-07T02:19:44.194191Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:19:44.194191Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TokSuite: Measuring the Impact of Tokenizer Choice on Language Model Behavior","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Brian Lester, Colin Raffel, Fengyuan Liu, G\\\"ul Sena Alt{\\i}nta\\c{s}, Malikeh Ehghaghi, Marco Ciccone, Wanru Zhao","submitted_at":"2025-12-23T20:43:06Z","abstract_excerpt":"Tokenizers provide the fundamental basis through which text is represented and processed by language models (LMs). Despite the importance of tokenization, its role in LM performance and behavior is poorly understood due to the challenge of measuring the impact of tokenization in isolation. To address this need, we present TokSuite, a collection of models and a benchmark that supports research into tokenization's influence on LMs. Specifically, we release fourteen pre-trained models that use different off-the-shelf tokenizers but are otherwise identical, using the same architecture, dataset, tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2512.20757","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2512.20757/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2512.20757","created_at":"2026-07-07T02:19:44.194285+00:00"},{"alias_kind":"arxiv_version","alias_value":"2512.20757v2","created_at":"2026-07-07T02:19:44.194285+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2512.20757","created_at":"2026-07-07T02:19:44.194285+00:00"},{"alias_kind":"pith_short_12","alias_value":"TO66H2RRGZC3","created_at":"2026-07-07T02:19:44.194285+00:00"},{"alias_kind":"pith_short_16","alias_value":"TO66H2RRGZC3T5BU","created_at":"2026-07-07T02:19:44.194285+00:00"},{"alias_kind":"pith_short_8","alias_value":"TO66H2RR","created_at":"2026-07-07T02:19:44.194285+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.01045","citing_title":"Child-directed speech facilitates production, not comprehension, in BabyLMs","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2604.16037","citing_title":"Stochasticity in Tokenisation Improves Robustness","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY","json":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY.json","graph_json":"https://pith.science/api/pith-number/TO66H2RRGZC3T5BUSORVT77PAY/graph.json","events_json":"https://pith.science/api/pith-number/TO66H2RRGZC3T5BUSORVT77PAY/events.json","paper":"https://pith.science/paper/TO66H2RR"},"agent_actions":{"view_html":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY","download_json":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY.json","view_paper":"https://pith.science/paper/TO66H2RR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2512.20757&json=true","fetch_graph":"https://pith.science/api/pith-number/TO66H2RRGZC3T5BUSORVT77PAY/graph.json","fetch_events":"https://pith.science/api/pith-number/TO66H2RRGZC3T5BUSORVT77PAY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY/action/storage_attestation","attest_author":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY/action/author_attestation","sign_citation":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY/action/citation_signature","submit_replication":"https://pith.science/pith/TO66H2RRGZC3T5BUSORVT77PAY/action/replication_record"}},"created_at":"2026-07-07T02:19:44.194285+00:00","updated_at":"2026-07-07T02:19:44.194285+00:00"}