{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KDLQGMOTUZSI6QBY35GFQBATCD","short_pith_number":"pith:KDLQGMOT","schema_version":"1.0","canonical_sha256":"50d70331d3a6648f4038df4c58041310f2c3f034603364f049bcaa6d537bfea0","source":{"kind":"arxiv","id":"2405.12981","version":1},"attestation_state":"computed","paper":{"title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aniruddha Nrusimha, Jonathan Ragan Kelly, Mayank Mishra, Rameswar Panda, William Brandon","submitted_at":"2024-05-21T17:59:29Z","abstract_excerpt":"Key-value (KV) caching plays an essential role in accelerating decoding for transformer-based autoregressive large language models (LLMs). However, the amount of memory required to store the KV cache can become prohibitive at long sequence lengths and large batch sizes. Since the invention of the transformer, two of the most effective interventions discovered for reducing the size of the KV cache have been Multi-Query Attention (MQA) and its generalization, Grouped-Query Attention (GQA). MQA and GQA both modify the design of the attention block so that multiple query heads can share a single k"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12981","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-21T17:59:29Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"38e542352699e41ef7bb08fd10f0eb6cf9d6288ed731580ebeac1be21853ee79","abstract_canon_sha256":"2633ba6d614454859b855dd36d921dd278ab6fe19d063aeb6005f4f13bba2e02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:21:29.343363Z","signature_b64":"AYjVYI4KkK+iUeaQzkD4LNCFXa6ALPG39nygDa38g1PfJx8WnPx+LIKAGsyUUk+Olb4Nq3lThpNLnjS061KrAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50d70331d3a6648f4038df4c58041310f2c3f034603364f049bcaa6d537bfea0","last_reissued_at":"2026-07-05T08:21:29.342890Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:21:29.342890Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Aniruddha Nrusimha, Jonathan Ragan Kelly, Mayank Mishra, Rameswar Panda, William Brandon","submitted_at":"2024-05-21T17:59:29Z","abstract_excerpt":"Key-value (KV) caching plays an essential role in accelerating decoding for transformer-based autoregressive large language models (LLMs). However, the amount of memory required to store the KV cache can become prohibitive at long sequence lengths and large batch sizes. Since the invention of the transformer, two of the most effective interventions discovered for reducing the size of the KV cache have been Multi-Query Attention (MQA) and its generalization, Grouped-Query Attention (GQA). MQA and GQA both modify the design of the attention block so that multiple query heads can share a single k"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12981","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12981","created_at":"2026-07-05T08:21:29.342942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12981v1","created_at":"2026-07-05T08:21:29.342942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12981","created_at":"2026-07-05T08:21:29.342942+00:00"},{"alias_kind":"pith_short_12","alias_value":"KDLQGMOTUZSI","created_at":"2026-07-05T08:21:29.342942+00:00"},{"alias_kind":"pith_short_16","alias_value":"KDLQGMOTUZSI6QBY","created_at":"2026-07-05T08:21:29.342942+00:00"},{"alias_kind":"pith_short_8","alias_value":"KDLQGMOT","created_at":"2026-07-05T08:21:29.342942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02780","citing_title":"Do Value Vectors in Deep Layers Need Context from the Residual Stream?","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13846","citing_title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01941","citing_title":"Semantic Integrity Matters: Benchmarking and Preserving High-Density Reasoning in KV Cache Compression","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07721","citing_title":"Memory-Efficient Looped Transformer: Decoupling Compute from Memory in Looped Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22782","citing_title":"Stochastic KV Routing: Enabling Adaptive Depth-Wise Cache Sharing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07721","citing_title":"Memory-Efficient Looped Transformer: Decoupling Compute from Memory in Looped Language Models","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD","json":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD.json","graph_json":"https://pith.science/api/pith-number/KDLQGMOTUZSI6QBY35GFQBATCD/graph.json","events_json":"https://pith.science/api/pith-number/KDLQGMOTUZSI6QBY35GFQBATCD/events.json","paper":"https://pith.science/paper/KDLQGMOT"},"agent_actions":{"view_html":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD","download_json":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD.json","view_paper":"https://pith.science/paper/KDLQGMOT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12981&json=true","fetch_graph":"https://pith.science/api/pith-number/KDLQGMOTUZSI6QBY35GFQBATCD/graph.json","fetch_events":"https://pith.science/api/pith-number/KDLQGMOTUZSI6QBY35GFQBATCD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD/action/storage_attestation","attest_author":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD/action/author_attestation","sign_citation":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD/action/citation_signature","submit_replication":"https://pith.science/pith/KDLQGMOTUZSI6QBY35GFQBATCD/action/replication_record"}},"created_at":"2026-07-05T08:21:29.342942+00:00","updated_at":"2026-07-05T08:21:29.342942+00:00"}