{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:B37CJUVA2FKMLHA6WCPYJVFZP7","short_pith_number":"pith:B37CJUVA","schema_version":"1.0","canonical_sha256":"0efe24d2a0d154c59c1eb09f84d4b97fdc1fc4c008849a3c9415525c48652276","source":{"kind":"arxiv","id":"2106.06899","version":1},"attestation_state":"computed","paper":{"title":"Memory-efficient Transformers via Top-$k$ Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ankit Gupta, David Ciprut, Guy Dar, Jonathan Berant, Shaya Goodman","submitted_at":"2021-06-13T02:30:23Z","abstract_excerpt":"Following the success of dot-product attention in Transformers, numerous approximations have been recently proposed to address its quadratic complexity with respect to the input length. While these variants are memory and compute efficient, it is not possible to directly use them with popular pre-trained language models trained using vanilla attention, without an expensive corrective pre-training stage. In this work, we propose a simple yet highly accurate approximation for vanilla attention. We process the queries in chunks, and for each query, compute the top-$k$ scores with respect to the k"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.06899","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-06-13T02:30:23Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2eb9506753d790f1fdbcd19e9d9d692cfe7937759c0627d4d4b67e0e70812bcf","abstract_canon_sha256":"433b08da1f8fb955425b0ea9b09c6ea7fc2c215f42aaf679e640be9ce4a5fd10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:42.243569Z","signature_b64":"E746YmNxC0Ck9qYLw7CwXkqV8l9F6O2NGsutnErLrKjpPRtywWpRVz4h2KFe379NDD28SM2Wz49F5cPYD1JQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0efe24d2a0d154c59c1eb09f84d4b97fdc1fc4c008849a3c9415525c48652276","last_reissued_at":"2026-07-05T02:48:42.243156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:42.243156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memory-efficient Transformers via Top-$k$ Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ankit Gupta, David Ciprut, Guy Dar, Jonathan Berant, Shaya Goodman","submitted_at":"2021-06-13T02:30:23Z","abstract_excerpt":"Following the success of dot-product attention in Transformers, numerous approximations have been recently proposed to address its quadratic complexity with respect to the input length. While these variants are memory and compute efficient, it is not possible to directly use them with popular pre-trained language models trained using vanilla attention, without an expensive corrective pre-training stage. In this work, we propose a simple yet highly accurate approximation for vanilla attention. We process the queries in chunks, and for each query, compute the top-$k$ scores with respect to the k"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06899","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.06899/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.06899","created_at":"2026-07-05T02:48:42.243208+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.06899v1","created_at":"2026-07-05T02:48:42.243208+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06899","created_at":"2026-07-05T02:48:42.243208+00:00"},{"alias_kind":"pith_short_12","alias_value":"B37CJUVA2FKM","created_at":"2026-07-05T02:48:42.243208+00:00"},{"alias_kind":"pith_short_16","alias_value":"B37CJUVA2FKMLHA6","created_at":"2026-07-05T02:48:42.243208+00:00"},{"alias_kind":"pith_short_8","alias_value":"B37CJUVA","created_at":"2026-07-05T02:48:42.243208+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19893","citing_title":"SSV: Sparse Speculative Verification for Efficient LLM Inference","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10539","citing_title":"IceCache: Memory-efficient KV-cache Management for Long-Sequence LLMs","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7","json":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7.json","graph_json":"https://pith.science/api/pith-number/B37CJUVA2FKMLHA6WCPYJVFZP7/graph.json","events_json":"https://pith.science/api/pith-number/B37CJUVA2FKMLHA6WCPYJVFZP7/events.json","paper":"https://pith.science/paper/B37CJUVA"},"agent_actions":{"view_html":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7","download_json":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7.json","view_paper":"https://pith.science/paper/B37CJUVA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.06899&json=true","fetch_graph":"https://pith.science/api/pith-number/B37CJUVA2FKMLHA6WCPYJVFZP7/graph.json","fetch_events":"https://pith.science/api/pith-number/B37CJUVA2FKMLHA6WCPYJVFZP7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7/action/storage_attestation","attest_author":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7/action/author_attestation","sign_citation":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7/action/citation_signature","submit_replication":"https://pith.science/pith/B37CJUVA2FKMLHA6WCPYJVFZP7/action/replication_record"}},"created_at":"2026-07-05T02:48:42.243208+00:00","updated_at":"2026-07-05T02:48:42.243208+00:00"}