{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ODTCHLR2MLIRYQAIDTZON74J6V","short_pith_number":"pith:ODTCHLR2","schema_version":"1.0","canonical_sha256":"70e623ae3a62d11c40081cf2e6ff89f56438b4d1a429448896964e3997726be9","source":{"kind":"arxiv","id":"2204.08832","version":1},"attestation_state":"computed","paper":{"title":"Impact of Tokenization on Language Models: An Analysis for Turkish","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cagri Toraman, Eyup Halit Yilmaz, Furkan \\c{S}ahinu\\c{c}, Oguzhan Ozcelik","submitted_at":"2022-04-19T12:01:46Z","abstract_excerpt":"Tokenization is an important text preprocessing step to prepare input tokens for deep language models. WordPiece and BPE are de facto methods employed by important models, such as BERT and GPT. However, the impact of tokenization can be different for morphologically rich languages, such as Turkic languages, where many words can be generated by adding prefixes and suffixes. We compare five tokenizers at different granularity levels, i.e. their outputs vary from smallest pieces of characters to the surface form of words, including a Morphological-level tokenizer. We train these tokenizers and pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.08832","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-04-19T12:01:46Z","cross_cats_sorted":[],"title_canon_sha256":"9d61ea2f8abd52b13a9149c0ce8f4b295f718df3be2ffd3268811607d962e9ea","abstract_canon_sha256":"9e39d2d3acfa82fca1b0cf34619060e8a80c35d81a1918ce9c4ba9f1c7075475"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:54:28.788600Z","signature_b64":"PEOYopx8NBsFBB6hlnQxJB+fW0VzHPirHFvZX5xAkMRDnfmLModybg2q//r1nkrxL6+r9SIdsOQ39xc+gQ5qDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70e623ae3a62d11c40081cf2e6ff89f56438b4d1a429448896964e3997726be9","last_reissued_at":"2026-07-05T05:54:28.788063Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:54:28.788063Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Impact of Tokenization on Language Models: An Analysis for Turkish","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cagri Toraman, Eyup Halit Yilmaz, Furkan \\c{S}ahinu\\c{c}, Oguzhan Ozcelik","submitted_at":"2022-04-19T12:01:46Z","abstract_excerpt":"Tokenization is an important text preprocessing step to prepare input tokens for deep language models. WordPiece and BPE are de facto methods employed by important models, such as BERT and GPT. However, the impact of tokenization can be different for morphologically rich languages, such as Turkic languages, where many words can be generated by adding prefixes and suffixes. We compare five tokenizers at different granularity levels, i.e. their outputs vary from smallest pieces of characters to the surface form of words, including a Morphological-level tokenizer. We train these tokenizers and pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.08832","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.08832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.08832","created_at":"2026-07-05T05:54:28.788124+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.08832v1","created_at":"2026-07-05T05:54:28.788124+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.08832","created_at":"2026-07-05T05:54:28.788124+00:00"},{"alias_kind":"pith_short_12","alias_value":"ODTCHLR2MLIR","created_at":"2026-07-05T05:54:28.788124+00:00"},{"alias_kind":"pith_short_16","alias_value":"ODTCHLR2MLIRYQAI","created_at":"2026-07-05T05:54:28.788124+00:00"},{"alias_kind":"pith_short_8","alias_value":"ODTCHLR2","created_at":"2026-07-05T05:54:28.788124+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04825","citing_title":"Plausibility as Commonsense Reasoning: Humans Succeed, Large Language Models Do not","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V","json":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V.json","graph_json":"https://pith.science/api/pith-number/ODTCHLR2MLIRYQAIDTZON74J6V/graph.json","events_json":"https://pith.science/api/pith-number/ODTCHLR2MLIRYQAIDTZON74J6V/events.json","paper":"https://pith.science/paper/ODTCHLR2"},"agent_actions":{"view_html":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V","download_json":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V.json","view_paper":"https://pith.science/paper/ODTCHLR2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.08832&json=true","fetch_graph":"https://pith.science/api/pith-number/ODTCHLR2MLIRYQAIDTZON74J6V/graph.json","fetch_events":"https://pith.science/api/pith-number/ODTCHLR2MLIRYQAIDTZON74J6V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V/action/storage_attestation","attest_author":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V/action/author_attestation","sign_citation":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V/action/citation_signature","submit_replication":"https://pith.science/pith/ODTCHLR2MLIRYQAIDTZON74J6V/action/replication_record"}},"created_at":"2026-07-05T05:54:28.788124+00:00","updated_at":"2026-07-05T05:54:28.788124+00:00"}