{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:R52IIMCFNFL4MTNIEE7HRC2DXK","short_pith_number":"pith:R52IIMCF","schema_version":"1.0","canonical_sha256":"8f748430456957c64da8213e788b43ba99417b7753929feb3c853271149e8e4c","source":{"kind":"arxiv","id":"2412.06538","version":1},"attestation_state":"computed","paper":{"title":"Understanding Factual Recall in Transformers via Associative Memories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alberto Bietti, Eshaan Nichani, Jason D. Lee","submitted_at":"2024-12-09T14:48:14Z","abstract_excerpt":"Large language models have demonstrated an impressive ability to perform factual recall. Prior work has found that transformers trained on factual recall tasks can store information at a rate proportional to their parameter count. In our work, we show that shallow transformers can use a combination of associative memories to obtain such near optimal storage capacity. We begin by proving that the storage capacities of both linear and MLP associative memories scale linearly with parameter count. We next introduce a synthetic factual recall task, and prove that a transformer with a single layer o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06538","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-09T14:48:14Z","cross_cats_sorted":["cs.CL","cs.IT","math.IT","stat.ML"],"title_canon_sha256":"6ec5c7930b751a9928e55e42643092faca0fed6c284db71a6f789e9a9ff22114","abstract_canon_sha256":"5561eca332c2732fa3835225afb10a73a5fe4b2ce287b924b1c6b0ddb91607c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:33.667606Z","signature_b64":"5PZXYfh1SZtj1hb748WmIDZLlBB9T0paCDkZOAo9jtClCZsn7uoNjdIySl/ncIjeTvQ4W/N03AHcRSqd9iTXCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f748430456957c64da8213e788b43ba99417b7753929feb3c853271149e8e4c","last_reissued_at":"2026-07-05T09:46:33.667106Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:33.667106Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Factual Recall in Transformers via Associative Memories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alberto Bietti, Eshaan Nichani, Jason D. Lee","submitted_at":"2024-12-09T14:48:14Z","abstract_excerpt":"Large language models have demonstrated an impressive ability to perform factual recall. Prior work has found that transformers trained on factual recall tasks can store information at a rate proportional to their parameter count. In our work, we show that shallow transformers can use a combination of associative memories to obtain such near optimal storage capacity. We begin by proving that the storage capacities of both linear and MLP associative memories scale linearly with parameter count. We next introduce a synthetic factual recall task, and prove that a transformer with a single layer o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06538","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06538/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06538","created_at":"2026-07-05T09:46:33.667161+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06538v1","created_at":"2026-07-05T09:46:33.667161+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06538","created_at":"2026-07-05T09:46:33.667161+00:00"},{"alias_kind":"pith_short_12","alias_value":"R52IIMCFNFL4","created_at":"2026-07-05T09:46:33.667161+00:00"},{"alias_kind":"pith_short_16","alias_value":"R52IIMCFNFL4MTNI","created_at":"2026-07-05T09:46:33.667161+00:00"},{"alias_kind":"pith_short_8","alias_value":"R52IIMCF","created_at":"2026-07-05T09:46:33.667161+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":87,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04662","citing_title":"Why Muon Outperforms Adam: A Curvature Perspective","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00901","citing_title":"Provable Knowledge Acquisition and Extraction in One-Layer Transformers","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26745","citing_title":"Deep sequence models tend to memorize geometrically; it is unclear why","ref_index":130,"is_internal_anchor":false},{"citing_arxiv_id":"2603.26554","citing_title":"Sharp Capacity Scaling of Spectral Optimizers in Learning Associative Memory","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06216","citing_title":"TIDE: Every Layer Knows the Token Beneath the Context","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05189","citing_title":"Sharp Capacity Thresholds in Linear Associative Memory: From Winner-Take-All to Listwise Retrieval","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK","json":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK.json","graph_json":"https://pith.science/api/pith-number/R52IIMCFNFL4MTNIEE7HRC2DXK/graph.json","events_json":"https://pith.science/api/pith-number/R52IIMCFNFL4MTNIEE7HRC2DXK/events.json","paper":"https://pith.science/paper/R52IIMCF"},"agent_actions":{"view_html":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK","download_json":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK.json","view_paper":"https://pith.science/paper/R52IIMCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06538&json=true","fetch_graph":"https://pith.science/api/pith-number/R52IIMCFNFL4MTNIEE7HRC2DXK/graph.json","fetch_events":"https://pith.science/api/pith-number/R52IIMCFNFL4MTNIEE7HRC2DXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK/action/storage_attestation","attest_author":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK/action/author_attestation","sign_citation":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK/action/citation_signature","submit_replication":"https://pith.science/pith/R52IIMCFNFL4MTNIEE7HRC2DXK/action/replication_record"}},"created_at":"2026-07-05T09:46:33.667161+00:00","updated_at":"2026-07-05T09:46:33.667161+00:00"}