{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CMNQRXSCTUKTWEWUFTDIVWXHGX","short_pith_number":"pith:CMNQRXSC","schema_version":"1.0","canonical_sha256":"131b08de429d153b12d42cc68adae735c1926359fefd50f69ad04e694c1fac44","source":{"kind":"arxiv","id":"2411.08671","version":1},"attestation_state":"computed","paper":{"title":"Theoretical Analysis of Byte-Pair Encoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.DS","authors_text":"Johannes Voderholzer, L\\'aszl\\'o Kozma","submitted_at":"2024-11-13T15:04:02Z","abstract_excerpt":"Byte-Pair Encoding (BPE) is a widely used method for subword tokenization, with origins in grammar-based text compression. It is employed in a variety of language processing tasks such as machine translation or large language model (LLM) pretraining, to create a token dictionary of a prescribed size. Most evaluations of BPE to date are empirical, and the reasons for its good practical performance are not well understood.\n  In this paper we focus on the optimization problem underlying BPE: finding a pair encoding that achieves optimal compression utility. We show that this problem is APX-comple"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.08671","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DS","submitted_at":"2024-11-13T15:04:02Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a4e806cd9500865b63a90e5142fb6221e8afc288750b8b65a268728c38d4506c","abstract_canon_sha256":"0890d92a6fbfa48a0b22803d745ec8779a364b79f7b9835a51c97b77adcb4db4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:00.653792Z","signature_b64":"pMz6uOpGEqJNyFUmqFytsLWcpZr0lmSlxquUp+ClSjxY35MKQiV9BfOzhoAS+sUFripl2hZY2e0QGNsBavDjBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"131b08de429d153b12d42cc68adae735c1926359fefd50f69ad04e694c1fac44","last_reissued_at":"2026-07-05T09:35:00.653302Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:00.653302Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Theoretical Analysis of Byte-Pair Encoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.DS","authors_text":"Johannes Voderholzer, L\\'aszl\\'o Kozma","submitted_at":"2024-11-13T15:04:02Z","abstract_excerpt":"Byte-Pair Encoding (BPE) is a widely used method for subword tokenization, with origins in grammar-based text compression. It is employed in a variety of language processing tasks such as machine translation or large language model (LLM) pretraining, to create a token dictionary of a prescribed size. Most evaluations of BPE to date are empirical, and the reasons for its good practical performance are not well understood.\n  In this paper we focus on the optimization problem underlying BPE: finding a pair encoding that achieves optimal compression utility. We show that this problem is APX-comple"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.08671","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.08671/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.08671","created_at":"2026-07-05T09:35:00.653366+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.08671v1","created_at":"2026-07-05T09:35:00.653366+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.08671","created_at":"2026-07-05T09:35:00.653366+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMNQRXSCTUKT","created_at":"2026-07-05T09:35:00.653366+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMNQRXSCTUKTWEWU","created_at":"2026-07-05T09:35:00.653366+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMNQRXSC","created_at":"2026-07-05T09:35:00.653366+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31509","citing_title":"Skill Reuse as Compression in Agentic RL","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02926","citing_title":"A Multi-head-based architecture for effective morphological tagging in Russian with open dictionary","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17814","citing_title":"Understanding Secret Leakage Risks in Code LLMs: A Tokenization Perspective","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX","json":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX.json","graph_json":"https://pith.science/api/pith-number/CMNQRXSCTUKTWEWUFTDIVWXHGX/graph.json","events_json":"https://pith.science/api/pith-number/CMNQRXSCTUKTWEWUFTDIVWXHGX/events.json","paper":"https://pith.science/paper/CMNQRXSC"},"agent_actions":{"view_html":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX","download_json":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX.json","view_paper":"https://pith.science/paper/CMNQRXSC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.08671&json=true","fetch_graph":"https://pith.science/api/pith-number/CMNQRXSCTUKTWEWUFTDIVWXHGX/graph.json","fetch_events":"https://pith.science/api/pith-number/CMNQRXSCTUKTWEWUFTDIVWXHGX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX/action/storage_attestation","attest_author":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX/action/author_attestation","sign_citation":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX/action/citation_signature","submit_replication":"https://pith.science/pith/CMNQRXSCTUKTWEWUFTDIVWXHGX/action/replication_record"}},"created_at":"2026-07-05T09:35:00.653366+00:00","updated_at":"2026-07-05T09:35:00.653366+00:00"}