{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:2PD2Q6VTZOI4TKAPZDU7FEK4UH","short_pith_number":"pith:2PD2Q6VT","schema_version":"1.0","canonical_sha256":"d3c7a87ab3cb91c9a80fc8e9f2915ca1d82ead9fc11a1f1713909bedf50a342b","source":{"kind":"arxiv","id":"2110.02782","version":2},"attestation_state":"computed","paper":{"title":"How BPE Affects Memorization in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dieuwke Hupkes, Eugene Kharitonov, Marco Baroni","submitted_at":"2021-10-06T14:01:56Z","abstract_excerpt":"Training data memorization in NLP can both be beneficial (e.g., closed-book QA) and undesirable (personal data extraction). In any case, successful model training requires a non-trivial amount of memorization to store word spellings, various linguistic idiosyncrasies and common knowledge. However, little is known about what affects the memorization behavior of NLP models, as the field tends to focus on the equally important question of generalization. In this work, we demonstrate that the size of the subword vocabulary learned by Byte-Pair Encoding (BPE) greatly affects both ability and tenden"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.02782","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-10-06T14:01:56Z","cross_cats_sorted":[],"title_canon_sha256":"e41ce9e521d1f328d7cc187bc6020bba66e8f3f8c17e8a4e373316cf6cf8c517","abstract_canon_sha256":"8e832b35d2a0d582f618e7a687536f692bc7ba609b6869c669bc57202f9415ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:36:57.449549Z","signature_b64":"hIeMhIF3AuA6b0gxzIgFHoXHCUx/1TjygkN4yePx5UTSddam/p57oZSjqldexjp1BUGc4gjTa5aT0POgYQMQDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3c7a87ab3cb91c9a80fc8e9f2915ca1d82ead9fc11a1f1713909bedf50a342b","last_reissued_at":"2026-07-05T03:36:57.448889Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:36:57.448889Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How BPE Affects Memorization in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dieuwke Hupkes, Eugene Kharitonov, Marco Baroni","submitted_at":"2021-10-06T14:01:56Z","abstract_excerpt":"Training data memorization in NLP can both be beneficial (e.g., closed-book QA) and undesirable (personal data extraction). In any case, successful model training requires a non-trivial amount of memorization to store word spellings, various linguistic idiosyncrasies and common knowledge. However, little is known about what affects the memorization behavior of NLP models, as the field tends to focus on the equally important question of generalization. In this work, we demonstrate that the size of the subword vocabulary learned by Byte-Pair Encoding (BPE) greatly affects both ability and tenden"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.02782","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.02782/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.02782","created_at":"2026-07-05T03:36:57.448950+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.02782v2","created_at":"2026-07-05T03:36:57.448950+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.02782","created_at":"2026-07-05T03:36:57.448950+00:00"},{"alias_kind":"pith_short_12","alias_value":"2PD2Q6VTZOI4","created_at":"2026-07-05T03:36:57.448950+00:00"},{"alias_kind":"pith_short_16","alias_value":"2PD2Q6VTZOI4TKAP","created_at":"2026-07-05T03:36:57.448950+00:00"},{"alias_kind":"pith_short_8","alias_value":"2PD2Q6VT","created_at":"2026-07-05T03:36:57.448950+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.00675","citing_title":"Speaker-Conditioned Phrase Break Prediction for Text-to-Speech with Phoneme-Level Pre-trained Language Model","ref_index":55,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH","json":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH.json","graph_json":"https://pith.science/api/pith-number/2PD2Q6VTZOI4TKAPZDU7FEK4UH/graph.json","events_json":"https://pith.science/api/pith-number/2PD2Q6VTZOI4TKAPZDU7FEK4UH/events.json","paper":"https://pith.science/paper/2PD2Q6VT"},"agent_actions":{"view_html":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH","download_json":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH.json","view_paper":"https://pith.science/paper/2PD2Q6VT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.02782&json=true","fetch_graph":"https://pith.science/api/pith-number/2PD2Q6VTZOI4TKAPZDU7FEK4UH/graph.json","fetch_events":"https://pith.science/api/pith-number/2PD2Q6VTZOI4TKAPZDU7FEK4UH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH/action/storage_attestation","attest_author":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH/action/author_attestation","sign_citation":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH/action/citation_signature","submit_replication":"https://pith.science/pith/2PD2Q6VTZOI4TKAPZDU7FEK4UH/action/replication_record"}},"created_at":"2026-07-05T03:36:57.448950+00:00","updated_at":"2026-07-05T03:36:57.448950+00:00"}