{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SZEZ7YZZLYDF36V7MX23ON6LUZ","short_pith_number":"pith:SZEZ7YZZ","schema_version":"1.0","canonical_sha256":"96499fe3395e065dfabf65f5b737cba664f571be1f2b6219b7eb1a48f9497aad","source":{"kind":"arxiv","id":"2503.02174","version":2},"attestation_state":"computed","paper":{"title":"Adversarial Tokenization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Guy Van den Broeck, Renato Lui Geh, Zilei Shao","submitted_at":"2025-03-04T01:31:17Z","abstract_excerpt":"Current LLM pipelines account for only one possible tokenization for a given string, ignoring exponentially many alternative tokenizations during training and inference. For example, the standard Llama3 tokenization of penguin is [p,enguin], yet [peng,uin] is another perfectly valid alternative. In this paper, we show that despite LLMs being trained solely on one tokenization, they still retain semantic understanding of other tokenizations, raising questions about their implications in LLM safety. Put succinctly, we answer the following question: can we adversarially tokenize an obviously mali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.02174","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-04T01:31:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f4dfde17ca8ed367f3eacc42d89ae4176556672566bb8859d83fc00df643d7a4","abstract_canon_sha256":"03d1f85fecb21dac684fe301a441aef973c58d6240d901f3ae791b506fac0079"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:56.331213Z","signature_b64":"i1GMjAo0wPsXI/oNLR1ypFHYM1NK99f534qYlXhgz9SfvcRFziurlf/mZzo3RHbpuYH6pGj2y6DWHFSifTurDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96499fe3395e065dfabf65f5b737cba664f571be1f2b6219b7eb1a48f9497aad","last_reissued_at":"2026-07-05T11:16:56.330652Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:56.330652Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adversarial Tokenization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Guy Van den Broeck, Renato Lui Geh, Zilei Shao","submitted_at":"2025-03-04T01:31:17Z","abstract_excerpt":"Current LLM pipelines account for only one possible tokenization for a given string, ignoring exponentially many alternative tokenizations during training and inference. For example, the standard Llama3 tokenization of penguin is [p,enguin], yet [peng,uin] is another perfectly valid alternative. In this paper, we show that despite LLMs being trained solely on one tokenization, they still retain semantic understanding of other tokenizations, raising questions about their implications in LLM safety. Put succinctly, we answer the following question: can we adversarially tokenize an obviously mali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02174","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.02174/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.02174","created_at":"2026-07-05T11:16:56.330727+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.02174v2","created_at":"2026-07-05T11:16:56.330727+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02174","created_at":"2026-07-05T11:16:56.330727+00:00"},{"alias_kind":"pith_short_12","alias_value":"SZEZ7YZZLYDF","created_at":"2026-07-05T11:16:56.330727+00:00"},{"alias_kind":"pith_short_16","alias_value":"SZEZ7YZZLYDF36V7","created_at":"2026-07-05T11:16:56.330727+00:00"},{"alias_kind":"pith_short_8","alias_value":"SZEZ7YZZ","created_at":"2026-07-05T11:16:56.330727+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.14122","citing_title":"Beyond Perplexity: UTF-8 Validity in Byte-aware Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ","json":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ.json","graph_json":"https://pith.science/api/pith-number/SZEZ7YZZLYDF36V7MX23ON6LUZ/graph.json","events_json":"https://pith.science/api/pith-number/SZEZ7YZZLYDF36V7MX23ON6LUZ/events.json","paper":"https://pith.science/paper/SZEZ7YZZ"},"agent_actions":{"view_html":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ","download_json":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ.json","view_paper":"https://pith.science/paper/SZEZ7YZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.02174&json=true","fetch_graph":"https://pith.science/api/pith-number/SZEZ7YZZLYDF36V7MX23ON6LUZ/graph.json","fetch_events":"https://pith.science/api/pith-number/SZEZ7YZZLYDF36V7MX23ON6LUZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ/action/storage_attestation","attest_author":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ/action/author_attestation","sign_citation":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ/action/citation_signature","submit_replication":"https://pith.science/pith/SZEZ7YZZLYDF36V7MX23ON6LUZ/action/replication_record"}},"created_at":"2026-07-05T11:16:56.330727+00:00","updated_at":"2026-07-05T11:16:56.330727+00:00"}