{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:X2WS3YD5ACZYJ4SMSKW2HOOYTJ","short_pith_number":"pith:X2WS3YD5","schema_version":"1.0","canonical_sha256":"bead2de07d00b384f24c92ada3b9d89a50c8250f06d92de40c0b983ddc93a5e7","source":{"kind":"arxiv","id":"2309.08715","version":1},"attestation_state":"computed","paper":{"title":"Formalizing BPE Tokenization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.FL","authors_text":"Brink van der Merwe (Stellenbosch University), Martin Berglund (Ume{\\aa} University)","submitted_at":"2023-09-15T19:10:42Z","abstract_excerpt":"In  this paper, we formalize practical byte pair encoding tokenization as it is used in large language models and other NLP systems, in particular we formally define and investigate the semantics of the SentencePiece and HuggingFace tokenizers, in particular how they relate to each other, depending on how the tokenization rules are constructed. Beyond this we consider how tokenization can be performed in an incremental fashion, as well as doing it left-to-right using an amount of memory constant in the length of the string, enabling e.g. using a finite state string-to-string transducer."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.08715","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.FL","submitted_at":"2023-09-15T19:10:42Z","cross_cats_sorted":[],"title_canon_sha256":"ef1bee7090342b5577f5b308a91c4a3bf96d2f82ae668b74140334d1c60a9e5a","abstract_canon_sha256":"944088d7ed8bda71929a21f658f2604d29f55764c58e77881b1ec93a1f9fd586"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:51:12.173652Z","signature_b64":"y/0O6BWpOLo4b964ay4kN8iGntJQHpy7iuoKBRkgH3EBvSN/ggNqjw5aPltDo2VWCTq002m/EL9n4jQP5qiACg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bead2de07d00b384f24c92ada3b9d89a50c8250f06d92de40c0b983ddc93a5e7","last_reissued_at":"2026-07-05T06:51:12.173243Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:51:12.173243Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Formalizing BPE Tokenization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.FL","authors_text":"Brink van der Merwe (Stellenbosch University), Martin Berglund (Ume{\\aa} University)","submitted_at":"2023-09-15T19:10:42Z","abstract_excerpt":"In  this paper, we formalize practical byte pair encoding tokenization as it is used in large language models and other NLP systems, in particular we formally define and investigate the semantics of the SentencePiece and HuggingFace tokenizers, in particular how they relate to each other, depending on how the tokenization rules are constructed. Beyond this we consider how tokenization can be performed in an incremental fashion, as well as doing it left-to-right using an amount of memory constant in the length of the string, enabling e.g. using a finite state string-to-string transducer."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.08715","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.08715/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.08715","created_at":"2026-07-05T06:51:12.173289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.08715v1","created_at":"2026-07-05T06:51:12.173289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.08715","created_at":"2026-07-05T06:51:12.173289+00:00"},{"alias_kind":"pith_short_12","alias_value":"X2WS3YD5ACZY","created_at":"2026-07-05T06:51:12.173289+00:00"},{"alias_kind":"pith_short_16","alias_value":"X2WS3YD5ACZYJ4SM","created_at":"2026-07-05T06:51:12.173289+00:00"},{"alias_kind":"pith_short_8","alias_value":"X2WS3YD5","created_at":"2026-07-05T06:51:12.173289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11015","citing_title":"DCVD: Dual-Channel Cross-Modal Fusion for Joint Vulnerability Detection and Localization","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ","json":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ.json","graph_json":"https://pith.science/api/pith-number/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/graph.json","events_json":"https://pith.science/api/pith-number/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/events.json","paper":"https://pith.science/paper/X2WS3YD5"},"agent_actions":{"view_html":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ","download_json":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ.json","view_paper":"https://pith.science/paper/X2WS3YD5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.08715&json=true","fetch_graph":"https://pith.science/api/pith-number/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/graph.json","fetch_events":"https://pith.science/api/pith-number/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/action/storage_attestation","attest_author":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/action/author_attestation","sign_citation":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/action/citation_signature","submit_replication":"https://pith.science/pith/X2WS3YD5ACZYJ4SMSKW2HOOYTJ/action/replication_record"}},"created_at":"2026-07-05T06:51:12.173289+00:00","updated_at":"2026-07-05T06:51:12.173289+00:00"}