{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:24HWZK72RLMGPIRD4A3M2MHK7F","short_pith_number":"pith:24HWZK72","schema_version":"1.0","canonical_sha256":"d70f6cabfa8ad867a223e036cd30eaf947d02253bb21b6f5fd93e27311693f41","source":{"kind":"arxiv","id":"2408.05646","version":2},"attestation_state":"computed","paper":{"title":"Eigen Attention: Attention in Low-Rank Space for KV Cache Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Gobinda Saha, Kaushik Roy, Sakshi Choudhary, Utkarsh Saxena","submitted_at":"2024-08-10T22:47:12Z","abstract_excerpt":"Large language models (LLMs) represent a groundbreaking advancement in the domain of natural language processing due to their impressive reasoning abilities. Recently, there has been considerable interest in increasing the context lengths for these models to enhance their applicability to complex tasks. However, at long context lengths and large batch sizes, the key-value (KV) cache, which stores the attention keys and values, emerges as the new bottleneck in memory usage during inference. To address this, we propose Eigen Attention, which performs the attention operation in a low-rank space, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.05646","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-10T22:47:12Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"9f542202e3aaf46b2026bb1a89dbd1ed7a724c21e17487bcffacad30b32aca71","abstract_canon_sha256":"7c919510d215404997efec1cd308d167b2c559ff0069513721af5adb507dc7fa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:47.550433Z","signature_b64":"WalTS/9WwJVRhIwOMfjyJJEKXeVIeQLrHbiQIuCElCbzB41tVDYzpqGnl9vbiXqtLfIA5CIxT9oHI/PY8qq+Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d70f6cabfa8ad867a223e036cd30eaf947d02253bb21b6f5fd93e27311693f41","last_reissued_at":"2026-07-05T09:32:47.549919Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:47.549919Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Eigen Attention: Attention in Low-Rank Space for KV Cache Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Gobinda Saha, Kaushik Roy, Sakshi Choudhary, Utkarsh Saxena","submitted_at":"2024-08-10T22:47:12Z","abstract_excerpt":"Large language models (LLMs) represent a groundbreaking advancement in the domain of natural language processing due to their impressive reasoning abilities. Recently, there has been considerable interest in increasing the context lengths for these models to enhance their applicability to complex tasks. However, at long context lengths and large batch sizes, the key-value (KV) cache, which stores the attention keys and values, emerges as the new bottleneck in memory usage during inference. To address this, we propose Eigen Attention, which performs the attention operation in a low-rank space, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.05646","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.05646/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.05646","created_at":"2026-07-05T09:32:47.549978+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.05646v2","created_at":"2026-07-05T09:32:47.549978+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.05646","created_at":"2026-07-05T09:32:47.549978+00:00"},{"alias_kind":"pith_short_12","alias_value":"24HWZK72RLMG","created_at":"2026-07-05T09:32:47.549978+00:00"},{"alias_kind":"pith_short_16","alias_value":"24HWZK72RLMGPIRD","created_at":"2026-07-05T09:32:47.549978+00:00"},{"alias_kind":"pith_short_8","alias_value":"24HWZK72","created_at":"2026-07-05T09:32:47.549978+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08565","citing_title":"EinSort: Sorting is All We Need for Tensorizing LLM","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08382","citing_title":"STAR-KV: Low-Rank KV Cache Compression via Soft Thresholding for Adaptive Rank Control","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16439","citing_title":"KVCapsule: Efficient Sequential KV Cache Compression for Vision-Language Models with Asymmetric Redundancy","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F","json":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F.json","graph_json":"https://pith.science/api/pith-number/24HWZK72RLMGPIRD4A3M2MHK7F/graph.json","events_json":"https://pith.science/api/pith-number/24HWZK72RLMGPIRD4A3M2MHK7F/events.json","paper":"https://pith.science/paper/24HWZK72"},"agent_actions":{"view_html":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F","download_json":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F.json","view_paper":"https://pith.science/paper/24HWZK72","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.05646&json=true","fetch_graph":"https://pith.science/api/pith-number/24HWZK72RLMGPIRD4A3M2MHK7F/graph.json","fetch_events":"https://pith.science/api/pith-number/24HWZK72RLMGPIRD4A3M2MHK7F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F/action/storage_attestation","attest_author":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F/action/author_attestation","sign_citation":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F/action/citation_signature","submit_replication":"https://pith.science/pith/24HWZK72RLMGPIRD4A3M2MHK7F/action/replication_record"}},"created_at":"2026-07-05T09:32:47.549978+00:00","updated_at":"2026-07-05T09:32:47.549978+00:00"}