{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PJ46MQBAPZUOTHJML4LJ46H4AP","short_pith_number":"pith:PJ46MQBA","schema_version":"1.0","canonical_sha256":"7a79e640207e68e99d2c5f169e78fc03d0ad20b513ac368d6af444b33309019b","source":{"kind":"arxiv","id":"2406.11214","version":3},"attestation_state":"computed","paper":{"title":"Problematic Tokens: Tokenizer Bias in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jin Yang, Yanbin Lin, Zhiqiang Wang, Zunduo Zhao","submitted_at":"2024-06-17T05:13:25Z","abstract_excerpt":"Recent advancements in large language models(LLMs), such as GPT-4 and GPT-4o, have shown exceptional performance, especially in languages with abundant resources like English, thanks to extensive datasets that ensure robust training. Conversely, these models exhibit limitations when processing under-resourced languages such as Chinese and Korean, where issues including hallucinatory responses remain prevalent. This paper traces the roots of these disparities to the tokenization process inherent to these models. Specifically, it explores how the tokenizers vocabulary, often used to speed up the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11214","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-17T05:13:25Z","cross_cats_sorted":[],"title_canon_sha256":"06536af1412fb6a9d685b5de529e51ab1a33a85b79348235f623c6dddb08ad74","abstract_canon_sha256":"6488332b52fbae82b66f86f9eb7978b5168c59dbb6b369fdf9f7015334afdf35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:10.613469Z","signature_b64":"nQ6VxoHkABsJnPgFouIax4RbEqQvCyjhzWYKLkJmh9zNkouyQjc6DqDazQ4liDmHZhRDEFUDrYT4pyiaS/+2Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a79e640207e68e99d2c5f169e78fc03d0ad20b513ac368d6af444b33309019b","last_reissued_at":"2026-07-05T09:35:10.613008Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:10.613008Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Problematic Tokens: Tokenizer Bias in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jin Yang, Yanbin Lin, Zhiqiang Wang, Zunduo Zhao","submitted_at":"2024-06-17T05:13:25Z","abstract_excerpt":"Recent advancements in large language models(LLMs), such as GPT-4 and GPT-4o, have shown exceptional performance, especially in languages with abundant resources like English, thanks to extensive datasets that ensure robust training. Conversely, these models exhibit limitations when processing under-resourced languages such as Chinese and Korean, where issues including hallucinatory responses remain prevalent. This paper traces the roots of these disparities to the tokenization process inherent to these models. Specifically, it explores how the tokenizers vocabulary, often used to speed up the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11214","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11214/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11214","created_at":"2026-07-05T09:35:10.613061+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11214v3","created_at":"2026-07-05T09:35:10.613061+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11214","created_at":"2026-07-05T09:35:10.613061+00:00"},{"alias_kind":"pith_short_12","alias_value":"PJ46MQBAPZUO","created_at":"2026-07-05T09:35:10.613061+00:00"},{"alias_kind":"pith_short_16","alias_value":"PJ46MQBAPZUOTHJM","created_at":"2026-07-05T09:35:10.613061+00:00"},{"alias_kind":"pith_short_8","alias_value":"PJ46MQBA","created_at":"2026-07-05T09:35:10.613061+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09630","citing_title":"Scratchpad Patching: Decoupling Compute from Patch Size in Byte-Level Language Models","ref_index":95,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP","json":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP.json","graph_json":"https://pith.science/api/pith-number/PJ46MQBAPZUOTHJML4LJ46H4AP/graph.json","events_json":"https://pith.science/api/pith-number/PJ46MQBAPZUOTHJML4LJ46H4AP/events.json","paper":"https://pith.science/paper/PJ46MQBA"},"agent_actions":{"view_html":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP","download_json":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP.json","view_paper":"https://pith.science/paper/PJ46MQBA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11214&json=true","fetch_graph":"https://pith.science/api/pith-number/PJ46MQBAPZUOTHJML4LJ46H4AP/graph.json","fetch_events":"https://pith.science/api/pith-number/PJ46MQBAPZUOTHJML4LJ46H4AP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP/action/storage_attestation","attest_author":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP/action/author_attestation","sign_citation":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP/action/citation_signature","submit_replication":"https://pith.science/pith/PJ46MQBAPZUOTHJML4LJ46H4AP/action/replication_record"}},"created_at":"2026-07-05T09:35:10.613061+00:00","updated_at":"2026-07-05T09:35:10.613061+00:00"}