{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:4RAZ23BFV7OBJC4WO4A2VTYCUK","short_pith_number":"pith:4RAZ23BF","schema_version":"1.0","canonical_sha256":"e4419d6c25afdc148b967701aacf02a2bae4cc23e2866c7dce46a7425aeaac10","source":{"kind":"arxiv","id":"2203.08913","version":1},"attestation_state":"computed","paper":{"title":"Memorizing Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Christian Szegedy, DeLesley Hutchins, Markus N. Rabe, Yuhuai Wu","submitted_at":"2022-03-16T19:54:35Z","abstract_excerpt":"Language models typically need to be trained or finetuned in order to acquire new knowledge, which involves updating their weights. We instead envision language models that can simply read and memorize new data at inference time, thus acquiring new knowledge immediately. In this work, we extend language models with the ability to memorize the internal representations of past inputs. We demonstrate that an approximate kNN lookup into a non-differentiable memory of recent (key, value) pairs improves language modeling across various benchmarks and tasks, including generic webtext (C4), math paper"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.08913","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-03-16T19:54:35Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"e42aefea62fc726a467c30c37f389de043facd106c367940ebbb7d647ff37321","abstract_canon_sha256":"e980aae04f60228bc69abc58a1f6360501f2d381d208b7b90fb4f0f9ef52dafc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:06:04.542651Z","signature_b64":"vJgiHKG420uKRnfza/pZlUZDPglaCX32gYLCwozfOyZmXYFr8lRlWBHJnhpQMfOxqqJFuNjBxRZHk/mJf/3ECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4419d6c25afdc148b967701aacf02a2bae4cc23e2866c7dce46a7425aeaac10","last_reissued_at":"2026-07-05T04:06:04.542156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:06:04.542156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memorizing Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Christian Szegedy, DeLesley Hutchins, Markus N. Rabe, Yuhuai Wu","submitted_at":"2022-03-16T19:54:35Z","abstract_excerpt":"Language models typically need to be trained or finetuned in order to acquire new knowledge, which involves updating their weights. We instead envision language models that can simply read and memorize new data at inference time, thus acquiring new knowledge immediately. In this work, we extend language models with the ability to memorize the internal representations of past inputs. We demonstrate that an approximate kNN lookup into a non-differentiable memory of recent (key, value) pairs improves language modeling across various benchmarks and tasks, including generic webtext (C4), math paper"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.08913","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.08913/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.08913","created_at":"2026-07-05T04:06:04.542231+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.08913v1","created_at":"2026-07-05T04:06:04.542231+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.08913","created_at":"2026-07-05T04:06:04.542231+00:00"},{"alias_kind":"pith_short_12","alias_value":"4RAZ23BFV7OB","created_at":"2026-07-05T04:06:04.542231+00:00"},{"alias_kind":"pith_short_16","alias_value":"4RAZ23BFV7OBJC4W","created_at":"2026-07-05T04:06:04.542231+00:00"},{"alias_kind":"pith_short_8","alias_value":"4RAZ23BF","created_at":"2026-07-05T04:06:04.542231+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":128,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20737","citing_title":"Repeated Shared Access Enables Grokking, but Edit Propagation Depends on an Addressable Memory","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02010","citing_title":"InduceKV: Fixed-Footprint Continual Adaptation of Multimodal LLMs via Inducing KV Memories","ref_index":203,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02303","citing_title":"A Hippocampus for Linear Attention: An Exact Memory for What the Recurrent State Forgets","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24051","citing_title":"Memento: Personalized RAG-Style Long-Retention Data Scaling for META Ads Recommendation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24930","citing_title":"H$^{2}$MT: Semantic Hierarchy-Aware Hierarchical Memory Transformer","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28876","citing_title":"Memory-Managed Long-Context Attention: Bounded Editable Memory with a Hard Lifecycle and Calibrated Sparse Fallback","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26797","citing_title":"Latent Recurrent Transformer: Architecture Exploration, Training Strategies, and Scaling Behavior","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22884","citing_title":"Tensor Cache: Eviction-conditioned Associative Memory for Transformers","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07143","citing_title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16893","citing_title":"NGM: A Plug-and-Play Training-Free Memory Module for LLMs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13674","citing_title":"PrefixMemory-Tuning: Modernizing Prefix-Tuning by Decoupling the Prefix from Attention","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12602","citing_title":"Exact Flow Linear Attention: Exact Solution from Continuous-Time Dynamics","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13370","citing_title":"Phasor Memory Networks: Stable Backpropagation Through Time for Scalable Explicit Memory","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10993","citing_title":"ECHO: Continuous Hierarchical Memory for Vision-Language-Action Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06225","citing_title":"Memory Inception: Latent-Space KV Cache Manipulation for Steering LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06225","citing_title":"Memory Inception: Latent-Space KV Cache Manipulation for Steering LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05066","citing_title":"The Impossibility Triangle of Long-Context Modeling","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10813","citing_title":"LongMemEval: Benchmarking Chat Assistants on Long-Term Interactive Memory","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2206.07682","citing_title":"Emergent Abilities of Large Language Models","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK","json":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK.json","graph_json":"https://pith.science/api/pith-number/4RAZ23BFV7OBJC4WO4A2VTYCUK/graph.json","events_json":"https://pith.science/api/pith-number/4RAZ23BFV7OBJC4WO4A2VTYCUK/events.json","paper":"https://pith.science/paper/4RAZ23BF"},"agent_actions":{"view_html":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK","download_json":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK.json","view_paper":"https://pith.science/paper/4RAZ23BF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.08913&json=true","fetch_graph":"https://pith.science/api/pith-number/4RAZ23BFV7OBJC4WO4A2VTYCUK/graph.json","fetch_events":"https://pith.science/api/pith-number/4RAZ23BFV7OBJC4WO4A2VTYCUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK/action/storage_attestation","attest_author":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK/action/author_attestation","sign_citation":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK/action/citation_signature","submit_replication":"https://pith.science/pith/4RAZ23BFV7OBJC4WO4A2VTYCUK/action/replication_record"}},"created_at":"2026-07-05T04:06:04.542231+00:00","updated_at":"2026-07-05T04:06:04.542231+00:00"}