{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EB5FDIVGKJQI7J7NQQBZK3B3TG","short_pith_number":"pith:EB5FDIVG","schema_version":"1.0","canonical_sha256":"207a51a2a652608fa7ed8403956c3b9993444ebb1e877c6cd98d86db0587606b","source":{"kind":"arxiv","id":"2506.03101","version":1},"attestation_state":"computed","paper":{"title":"Beyond Text Compression: Evaluating Tokenizers Across Scales","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ant\\'onio V. Lopes, Hendra Setiawan, Jonas F. Lotz, Leonardo Emili, Stephan Peitz","submitted_at":"2025-06-03T17:35:56Z","abstract_excerpt":"The choice of tokenizer can profoundly impact language model performance, yet accessible and reliable evaluations of tokenizer quality remain an open challenge. Inspired by scaling consistency, we show that smaller models can accurately predict significant differences in tokenizer impact on larger models at a fraction of the compute cost. By systematically evaluating both English-centric and multilingual tokenizers, we find that tokenizer choice has negligible effects on tasks in English but results in consistent performance differences in multilingual settings. We propose new intrinsic tokeni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03101","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-03T17:35:56Z","cross_cats_sorted":[],"title_canon_sha256":"1104674e1740cca8febc0581499605eaf2ef1f8ebc81f617911fc5e67544058e","abstract_canon_sha256":"3ee61e12ee34464d180fc33505a0782ae195f479a6fc530fb18eedb43150d954"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:14.888163Z","signature_b64":"M7eOzz6X9QpnYBo1N+hDBw4Lj8i1BWuC9qsqyqMOE3tKKm5IujWmUO8JqpLxDtXEA0iIXeSirJPAQaZAbKreDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"207a51a2a652608fa7ed8403956c3b9993444ebb1e877c6cd98d86db0587606b","last_reissued_at":"2026-07-05T11:15:14.887654Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:14.887654Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Text Compression: Evaluating Tokenizers Across Scales","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ant\\'onio V. Lopes, Hendra Setiawan, Jonas F. Lotz, Leonardo Emili, Stephan Peitz","submitted_at":"2025-06-03T17:35:56Z","abstract_excerpt":"The choice of tokenizer can profoundly impact language model performance, yet accessible and reliable evaluations of tokenizer quality remain an open challenge. Inspired by scaling consistency, we show that smaller models can accurately predict significant differences in tokenizer impact on larger models at a fraction of the compute cost. By systematically evaluating both English-centric and multilingual tokenizers, we find that tokenizer choice has negligible effects on tasks in English but results in consistent performance differences in multilingual settings. We propose new intrinsic tokeni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03101","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03101/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03101","created_at":"2026-07-05T11:15:14.887714+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03101v1","created_at":"2026-07-05T11:15:14.887714+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03101","created_at":"2026-07-05T11:15:14.887714+00:00"},{"alias_kind":"pith_short_12","alias_value":"EB5FDIVGKJQI","created_at":"2026-07-05T11:15:14.887714+00:00"},{"alias_kind":"pith_short_16","alias_value":"EB5FDIVGKJQI7J7N","created_at":"2026-07-05T11:15:14.887714+00:00"},{"alias_kind":"pith_short_8","alias_value":"EB5FDIVG","created_at":"2026-07-05T11:15:14.887714+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG","json":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG.json","graph_json":"https://pith.science/api/pith-number/EB5FDIVGKJQI7J7NQQBZK3B3TG/graph.json","events_json":"https://pith.science/api/pith-number/EB5FDIVGKJQI7J7NQQBZK3B3TG/events.json","paper":"https://pith.science/paper/EB5FDIVG"},"agent_actions":{"view_html":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG","download_json":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG.json","view_paper":"https://pith.science/paper/EB5FDIVG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03101&json=true","fetch_graph":"https://pith.science/api/pith-number/EB5FDIVGKJQI7J7NQQBZK3B3TG/graph.json","fetch_events":"https://pith.science/api/pith-number/EB5FDIVGKJQI7J7NQQBZK3B3TG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG/action/storage_attestation","attest_author":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG/action/author_attestation","sign_citation":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG/action/citation_signature","submit_replication":"https://pith.science/pith/EB5FDIVGKJQI7J7NQQBZK3B3TG/action/replication_record"}},"created_at":"2026-07-05T11:15:14.887714+00:00","updated_at":"2026-07-05T11:15:14.887714+00:00"}