{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:OHSEAYOVH6ED3QLELYLJUKSDST","short_pith_number":"pith:OHSEAYOV","schema_version":"1.0","canonical_sha256":"71e44061d53f883dc1645e169a2a4394ca7af0034d6288581189ad8342b9a3c3","source":{"kind":"arxiv","id":"2112.05682","version":3},"attestation_state":"computed","paper":{"title":"Self-attention Does Not Need $O(n^2)$ Memory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Charles Staats, Markus N. Rabe","submitted_at":"2021-12-10T17:25:07Z","abstract_excerpt":"We present a very simple algorithm for attention that requires $O(1)$ memory with respect to sequence length and an extension to self-attention that requires $O(\\log n)$ memory. This is in contrast with the frequently stated belief that self-attention requires $O(n^2)$ memory. While the time complexity is still $O(n^2)$, device memory rather than compute capability is often the limiting factor on modern accelerators. Thus, reducing the memory requirements of attention allows processing of longer sequences than might otherwise be feasible. We provide a practical implementation for accelerators "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.05682","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-10T17:25:07Z","cross_cats_sorted":[],"title_canon_sha256":"4a74d97c2cd794faace0b2c0a955a6d0351edbe12ce9cf67b2bc16343efea63c","abstract_canon_sha256":"35316aaeba440f33505455b8a078881aefec635db694106a894e67d7073e5fe8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:46.475691Z","signature_b64":"zSv3VxSy7ovFo/J5++GZYo0suIn8US5f2LvTbfvwRLO+Htv7gRk1xmkWn/jNMYHYMRJ6xHvn3qeXjKcfOiwLCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71e44061d53f883dc1645e169a2a4394ca7af0034d6288581189ad8342b9a3c3","last_reissued_at":"2026-07-05T05:04:46.475181Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:46.475181Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-attention Does Not Need $O(n^2)$ Memory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Charles Staats, Markus N. Rabe","submitted_at":"2021-12-10T17:25:07Z","abstract_excerpt":"We present a very simple algorithm for attention that requires $O(1)$ memory with respect to sequence length and an extension to self-attention that requires $O(\\log n)$ memory. This is in contrast with the frequently stated belief that self-attention requires $O(n^2)$ memory. While the time complexity is still $O(n^2)$, device memory rather than compute capability is often the limiting factor on modern accelerators. Thus, reducing the memory requirements of attention allows processing of longer sequences than might otherwise be feasible. We provide a practical implementation for accelerators "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.05682","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.05682/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.05682","created_at":"2026-07-05T05:04:46.475238+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.05682v3","created_at":"2026-07-05T05:04:46.475238+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.05682","created_at":"2026-07-05T05:04:46.475238+00:00"},{"alias_kind":"pith_short_12","alias_value":"OHSEAYOVH6ED","created_at":"2026-07-05T05:04:46.475238+00:00"},{"alias_kind":"pith_short_16","alias_value":"OHSEAYOVH6ED3QLE","created_at":"2026-07-05T05:04:46.475238+00:00"},{"alias_kind":"pith_short_8","alias_value":"OHSEAYOV","created_at":"2026-07-05T05:04:46.475238+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10537","citing_title":"Prefilling-dLLM: Predictive Prefilling for Long-Context Inference in Diffusion Language Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10305","citing_title":"Baichuan 2: Open Large-scale Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03594","citing_title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2411.12502","citing_title":"Transformer Neural Processes - Kernel Regression","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2407.08608","citing_title":"FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18226","citing_title":"Context Memorization for Efficient Long Context Generation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":128,"is_internal_anchor":false},{"citing_arxiv_id":"2408.00724","citing_title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05496","citing_title":"Flex Attention: A Programming Model for Generating Optimized Attention Kernels","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16867","citing_title":"The Falcon Series of Open Language Models","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00520","citing_title":"NEST: Nested Event Stream Transformer for Sequences of Multisets","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02654","citing_title":"Drift-Resilient Temporal Priors for Visual Tracking","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15408","citing_title":"Dispatch-Aware Ragged Attention for Pruned Vision Transformers","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2310.01889","citing_title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2205.14135","citing_title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23798","citing_title":"ELSA: Exact Linear-Scan Attention for Fast and Memory-Light Vision Transformers","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21215","citing_title":"The Recurrent Transformer: Greater Effective Depth and Efficient Decoding","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2307.08691","citing_title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15408","citing_title":"Dispatch-Aware Ragged Attention for Pruned Vision Transformers","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16864","citing_title":"HieraSparse: Hierarchical Semi-Structured Sparse KV Attention","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST","json":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST.json","graph_json":"https://pith.science/api/pith-number/OHSEAYOVH6ED3QLELYLJUKSDST/graph.json","events_json":"https://pith.science/api/pith-number/OHSEAYOVH6ED3QLELYLJUKSDST/events.json","paper":"https://pith.science/paper/OHSEAYOV"},"agent_actions":{"view_html":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST","download_json":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST.json","view_paper":"https://pith.science/paper/OHSEAYOV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.05682&json=true","fetch_graph":"https://pith.science/api/pith-number/OHSEAYOVH6ED3QLELYLJUKSDST/graph.json","fetch_events":"https://pith.science/api/pith-number/OHSEAYOVH6ED3QLELYLJUKSDST/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST/action/storage_attestation","attest_author":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST/action/author_attestation","sign_citation":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST/action/citation_signature","submit_replication":"https://pith.science/pith/OHSEAYOVH6ED3QLELYLJUKSDST/action/replication_record"}},"created_at":"2026-07-05T05:04:46.475238+00:00","updated_at":"2026-07-05T05:04:46.475238+00:00"}