{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:J2AD5BU2PFULNZS72S2R3XBQAM","short_pith_number":"pith:J2AD5BU2","schema_version":"1.0","canonical_sha256":"4e803e869a7968b6e65fd4b51ddc30033b8114e39897970819c9236623a405fb","source":{"kind":"arxiv","id":"2404.06508","version":3},"attestation_state":"computed","paper":{"title":"On the Effect of (Near) Duplicate Subwords in Language Modelling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anton Sch\\\"afer, Imanol Schlag, Thomas Hofmann, Tiago Pimentel","submitted_at":"2024-04-09T17:57:29Z","abstract_excerpt":"Tokenisation is a core part of language models (LMs). It involves splitting a character sequence into subwords which are assigned arbitrary indices before being served to the LM. While typically lossless, however, this process may lead to less sample efficient LM training: as it removes character-level information, it could make it harder for LMs to generalise across similar subwords, such as now and Now. We refer to such subwords as near duplicates. In this paper, we study the impact of near duplicate subwords on LM training efficiency. First, we design an experiment that gives us an upper bo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.06508","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-09T17:57:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d87c50cf79e2dbb7bc6bfa666934123974e24a5207e48958972620bfc21d069f","abstract_canon_sha256":"cf7572af735c4342e10fd3066bfd30a2c4b6fb140c45054caca1f7aa02ffe168"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:53.344636Z","signature_b64":"JAP+jlUoVZF+JVZTu7dwQTZG4GVybzScInQ2+LkstxQ2lTHHikEvVuxZpyE0lTfLXE6Pxdej61BsWsKXhphnDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e803e869a7968b6e65fd4b51ddc30033b8114e39897970819c9236623a405fb","last_reissued_at":"2026-07-05T08:44:53.344076Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:53.344076Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Effect of (Near) Duplicate Subwords in Language Modelling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anton Sch\\\"afer, Imanol Schlag, Thomas Hofmann, Tiago Pimentel","submitted_at":"2024-04-09T17:57:29Z","abstract_excerpt":"Tokenisation is a core part of language models (LMs). It involves splitting a character sequence into subwords which are assigned arbitrary indices before being served to the LM. While typically lossless, however, this process may lead to less sample efficient LM training: as it removes character-level information, it could make it harder for LMs to generalise across similar subwords, such as now and Now. We refer to such subwords as near duplicates. In this paper, we study the impact of near duplicate subwords on LM training efficiency. First, we design an experiment that gives us an upper bo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.06508","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.06508/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.06508","created_at":"2026-07-05T08:44:53.344137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.06508v3","created_at":"2026-07-05T08:44:53.344137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.06508","created_at":"2026-07-05T08:44:53.344137+00:00"},{"alias_kind":"pith_short_12","alias_value":"J2AD5BU2PFUL","created_at":"2026-07-05T08:44:53.344137+00:00"},{"alias_kind":"pith_short_16","alias_value":"J2AD5BU2PFULNZS7","created_at":"2026-07-05T08:44:53.344137+00:00"},{"alias_kind":"pith_short_8","alias_value":"J2AD5BU2","created_at":"2026-07-05T08:44:53.344137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM","json":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM.json","graph_json":"https://pith.science/api/pith-number/J2AD5BU2PFULNZS72S2R3XBQAM/graph.json","events_json":"https://pith.science/api/pith-number/J2AD5BU2PFULNZS72S2R3XBQAM/events.json","paper":"https://pith.science/paper/J2AD5BU2"},"agent_actions":{"view_html":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM","download_json":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM.json","view_paper":"https://pith.science/paper/J2AD5BU2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.06508&json=true","fetch_graph":"https://pith.science/api/pith-number/J2AD5BU2PFULNZS72S2R3XBQAM/graph.json","fetch_events":"https://pith.science/api/pith-number/J2AD5BU2PFULNZS72S2R3XBQAM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM/action/storage_attestation","attest_author":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM/action/author_attestation","sign_citation":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM/action/citation_signature","submit_replication":"https://pith.science/pith/J2AD5BU2PFULNZS72S2R3XBQAM/action/replication_record"}},"created_at":"2026-07-05T08:44:53.344137+00:00","updated_at":"2026-07-05T08:44:53.344137+00:00"}