{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XRDMPDN35CM74LPWONCUIMD6WN","short_pith_number":"pith:XRDMPDN3","schema_version":"1.0","canonical_sha256":"bc46c78dbbe899fe2df6734544307eb369f39e2b06c9390deb847ecee83cb639","source":{"kind":"arxiv","id":"2112.10508","version":1},"attestation_state":"computed","paper":{"title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Arun Raja, Beno\\^it Sagot, Chenglei Si, Colin Raffel, Elizabeth Salesky, Manan Dey, Matthias Gall\\'e, Sabrina J. Mielke, Samson Tan, Wilson Y. Lee, Zaid Alyafeai","submitted_at":"2021-12-20T13:04:18Z","abstract_excerpt":"What are the units of text that we want to model? From bytes to multi-word expressions, text can be analyzed and generated at many granularities. Until recently, most natural language processing (NLP) models operated over words, treating those as discrete and atomic tokens, but starting with byte-pair encoding (BPE), subword-based approaches have become dominant in many areas, enabling small vocabularies while still allowing for fast inference. Is the end of the road character-level model or byte-level processing? In this survey, we connect several lines of work from the pre-neural and neural "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.10508","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-12-20T13:04:18Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cfc2e9d664aac3bba150360940b4e93b04cb44466959dd558922b9c5b5811c2f","abstract_canon_sha256":"d67fa28882069912077842f82a1f5f66ce88a5581199e0bf21ea72a28dd6ff7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:42:16.044382Z","signature_b64":"hOl+kVH3IZmOGrWaBlpP3+BL215yN7OA90fQwG2Ehu9PlZgFyH/l68I502qv9DcbRFaHjXZQaPuCesfHozGeCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc46c78dbbe899fe2df6734544307eb369f39e2b06c9390deb847ecee83cb639","last_reissued_at":"2026-07-05T03:42:16.043988Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:42:16.043988Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Between words and characters: A Brief History of Open-Vocabulary Modeling and Tokenization in NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Arun Raja, Beno\\^it Sagot, Chenglei Si, Colin Raffel, Elizabeth Salesky, Manan Dey, Matthias Gall\\'e, Sabrina J. Mielke, Samson Tan, Wilson Y. Lee, Zaid Alyafeai","submitted_at":"2021-12-20T13:04:18Z","abstract_excerpt":"What are the units of text that we want to model? From bytes to multi-word expressions, text can be analyzed and generated at many granularities. Until recently, most natural language processing (NLP) models operated over words, treating those as discrete and atomic tokens, but starting with byte-pair encoding (BPE), subword-based approaches have become dominant in many areas, enabling small vocabularies while still allowing for fast inference. Is the end of the road character-level model or byte-level processing? In this survey, we connect several lines of work from the pre-neural and neural "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.10508","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.10508/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.10508","created_at":"2026-07-05T03:42:16.044040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.10508v1","created_at":"2026-07-05T03:42:16.044040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.10508","created_at":"2026-07-05T03:42:16.044040+00:00"},{"alias_kind":"pith_short_12","alias_value":"XRDMPDN35CM7","created_at":"2026-07-05T03:42:16.044040+00:00"},{"alias_kind":"pith_short_16","alias_value":"XRDMPDN35CM74LPW","created_at":"2026-07-05T03:42:16.044040+00:00"},{"alias_kind":"pith_short_8","alias_value":"XRDMPDN3","created_at":"2026-07-05T03:42:16.044040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27019","citing_title":"MinGram: A Minimalist Unigram Tokenizer with High Compression and Competitive Morphological Alignment","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20993","citing_title":"Phonemes to the Rescue: Multilingual Tokenization Based on International Phonetic Alphabet","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08713","citing_title":"The price of incrementality in k-center clustering","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29459","citing_title":"Kronecker Embeddings: Byte-Level Structured Token Representations for Parameter-Efficient Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22403","citing_title":"Translating Signals to Languages for sEMG-Based Activity Recognition","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17152","citing_title":"Multilingual and Multimodal LLMs in the Wild: Building for Low-Resource Languages","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14171","citing_title":"Benchmarking Linguistic Adaptation in Comparable-Sized LLMs: A Study of Llama-3.1-8B, Mistral-7B-v0.1, and Qwen3-8B on Romanized Nepali","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17564","citing_title":"BloombergGPT: A Large Language Model for Finance","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2403.19887","citing_title":"Jamba: A Hybrid Transformer-Mamba Language Model","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":278,"is_internal_anchor":false},{"citing_arxiv_id":"2211.05100","citing_title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18423","citing_title":"BhashaSutra: A Task-Centric Unified Survey of Indian NLP Datasets, Corpora, and Resources","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN","json":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN.json","graph_json":"https://pith.science/api/pith-number/XRDMPDN35CM74LPWONCUIMD6WN/graph.json","events_json":"https://pith.science/api/pith-number/XRDMPDN35CM74LPWONCUIMD6WN/events.json","paper":"https://pith.science/paper/XRDMPDN3"},"agent_actions":{"view_html":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN","download_json":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN.json","view_paper":"https://pith.science/paper/XRDMPDN3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.10508&json=true","fetch_graph":"https://pith.science/api/pith-number/XRDMPDN35CM74LPWONCUIMD6WN/graph.json","fetch_events":"https://pith.science/api/pith-number/XRDMPDN35CM74LPWONCUIMD6WN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN/action/storage_attestation","attest_author":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN/action/author_attestation","sign_citation":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN/action/citation_signature","submit_replication":"https://pith.science/pith/XRDMPDN35CM74LPWONCUIMD6WN/action/replication_record"}},"created_at":"2026-07-05T03:42:16.044040+00:00","updated_at":"2026-07-05T03:42:16.044040+00:00"}