{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:D6EIRONNQ3HDV7MMQMTDGWAPUH","short_pith_number":"pith:D6EIRONN","schema_version":"1.0","canonical_sha256":"1f8888b9ad86ce3afd8c832633580fa1e9114109aeb8755d9ca61b1f7a56bf44","source":{"kind":"arxiv","id":"2306.02010","version":3},"attestation_state":"computed","paper":{"title":"Memorization Capacity of Multi-Head Attention in Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Christos Thrampoulidis, Renjie Liao, Sadegh Mahdavi","submitted_at":"2023-06-03T05:45:29Z","abstract_excerpt":"Transformers have become the go-to architecture for language and vision tasks, yet their theoretical properties, especially memorization capacity, remain elusive. This paper investigates the memorization abilities of multi-head attention mechanisms, examining how many example sequences they can memorize, as a function of the number of heads and sequence length. Motivated by experimental findings on vision transformers, we introduce novel assumptions about the linear independence of input data, distinct from the commonly used general-position assumption. Under these assumptions, we demonstrate "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.02010","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-03T05:45:29Z","cross_cats_sorted":[],"title_canon_sha256":"ed711aec10ad03067ded88513bffff085f5d1e430a47c0ec74f5afc306bfc5cc","abstract_canon_sha256":"9ad0b897d2c445b47420b11b2962fb78d25eafbac4a750f893049da0de033889"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:09.325615Z","signature_b64":"ZR6pUwZ6BvAZfaW8zcJHCMIdKEnlB+KraN/58DvNfQU6xWYbGscEqaw0wFEnhOLoUqlr+hws6qdiy9QBV9eXDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f8888b9ad86ce3afd8c832633580fa1e9114109aeb8755d9ca61b1f7a56bf44","last_reissued_at":"2026-07-05T07:51:09.325174Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:09.325174Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memorization Capacity of Multi-Head Attention in Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Christos Thrampoulidis, Renjie Liao, Sadegh Mahdavi","submitted_at":"2023-06-03T05:45:29Z","abstract_excerpt":"Transformers have become the go-to architecture for language and vision tasks, yet their theoretical properties, especially memorization capacity, remain elusive. This paper investigates the memorization abilities of multi-head attention mechanisms, examining how many example sequences they can memorize, as a function of the number of heads and sequence length. Motivated by experimental findings on vision transformers, we introduce novel assumptions about the linear independence of input data, distinct from the commonly used general-position assumption. Under these assumptions, we demonstrate "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02010","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02010/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.02010","created_at":"2026-07-05T07:51:09.325231+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.02010v3","created_at":"2026-07-05T07:51:09.325231+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02010","created_at":"2026-07-05T07:51:09.325231+00:00"},{"alias_kind":"pith_short_12","alias_value":"D6EIRONNQ3HD","created_at":"2026-07-05T07:51:09.325231+00:00"},{"alias_kind":"pith_short_16","alias_value":"D6EIRONNQ3HDV7MM","created_at":"2026-07-05T07:51:09.325231+00:00"},{"alias_kind":"pith_short_8","alias_value":"D6EIRONN","created_at":"2026-07-05T07:51:09.325231+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12546","citing_title":"Extracting memorized pieces of (copyrighted) books from open-weight language models","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00901","citing_title":"Provable Knowledge Acquisition and Extraction in One-Layer Transformers","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH","json":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH.json","graph_json":"https://pith.science/api/pith-number/D6EIRONNQ3HDV7MMQMTDGWAPUH/graph.json","events_json":"https://pith.science/api/pith-number/D6EIRONNQ3HDV7MMQMTDGWAPUH/events.json","paper":"https://pith.science/paper/D6EIRONN"},"agent_actions":{"view_html":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH","download_json":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH.json","view_paper":"https://pith.science/paper/D6EIRONN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.02010&json=true","fetch_graph":"https://pith.science/api/pith-number/D6EIRONNQ3HDV7MMQMTDGWAPUH/graph.json","fetch_events":"https://pith.science/api/pith-number/D6EIRONNQ3HDV7MMQMTDGWAPUH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH/action/storage_attestation","attest_author":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH/action/author_attestation","sign_citation":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH/action/citation_signature","submit_replication":"https://pith.science/pith/D6EIRONNQ3HDV7MMQMTDGWAPUH/action/replication_record"}},"created_at":"2026-07-05T07:51:09.325231+00:00","updated_at":"2026-07-05T07:51:09.325231+00:00"}