{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XKEB66JJVS5LCBOEW67GYC5PC5","short_pith_number":"pith:XKEB66JJ","schema_version":"1.0","canonical_sha256":"ba881f7929acbab105c4b7be6c0baf174425455bf3a87d62a31d1aa42aa81844","source":{"kind":"arxiv","id":"2404.04793","version":2},"attestation_state":"computed","paper":{"title":"SqueezeAttention: 2D Management of KV-Cache in LLM Inference via Layer-wise Optimal Budget","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bin Cui, Shaoduo Gan, Zihao Wang","submitted_at":"2024-04-07T03:08:14Z","abstract_excerpt":"Optimizing the Key-Value (KV) cache of the Large Language Model (LLM) has been considered critical to saving the cost of inference. Most of the existing KV-cache compression algorithms attempted to sparsify the sequence of tokens by taking advantage of the different importance of tokens. However, most of these methods treat all layers equally, allocating the same KV budget to each layer. This approach is suboptimal, as some layers may be less sensitive to input tokens yet still receive the same budget as others. In this work, we found that by identifying the importance of attention layers, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.04793","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-04-07T03:08:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8229b36148f318fc916fd5dc7078b1549c491778d9cdb35ad761b3bbc82397bd","abstract_canon_sha256":"2ad65b7937757041f562c7739ebc8ef2f3e6b6f07a6fca597fa9814a416c8527"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:24.514152Z","signature_b64":"BCBiU8nCFp2aoncTIBe/kXYeoklZ17kJwfZQNmY4cSFAOvCdAybm+/cEUX065IKvV0+y98Bfb/+mpYj25tcLDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba881f7929acbab105c4b7be6c0baf174425455bf3a87d62a31d1aa42aa81844","last_reissued_at":"2026-07-05T09:18:24.513669Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:24.513669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SqueezeAttention: 2D Management of KV-Cache in LLM Inference via Layer-wise Optimal Budget","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Bin Cui, Shaoduo Gan, Zihao Wang","submitted_at":"2024-04-07T03:08:14Z","abstract_excerpt":"Optimizing the Key-Value (KV) cache of the Large Language Model (LLM) has been considered critical to saving the cost of inference. Most of the existing KV-cache compression algorithms attempted to sparsify the sequence of tokens by taking advantage of the different importance of tokens. However, most of these methods treat all layers equally, allocating the same KV budget to each layer. This approach is suboptimal, as some layers may be less sensitive to input tokens yet still receive the same budget as others. In this work, we found that by identifying the importance of attention layers, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.04793","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.04793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.04793","created_at":"2026-07-05T09:18:24.513729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.04793v2","created_at":"2026-07-05T09:18:24.513729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.04793","created_at":"2026-07-05T09:18:24.513729+00:00"},{"alias_kind":"pith_short_12","alias_value":"XKEB66JJVS5L","created_at":"2026-07-05T09:18:24.513729+00:00"},{"alias_kind":"pith_short_16","alias_value":"XKEB66JJVS5LCBOE","created_at":"2026-07-05T09:18:24.513729+00:00"},{"alias_kind":"pith_short_8","alias_value":"XKEB66JJ","created_at":"2026-07-05T09:18:24.513729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.02262","citing_title":"WindowQuant: Mixed-Precision KV Cache Quantization based on Window-Level Similarity for VLMs Inference Optimization","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5","json":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5.json","graph_json":"https://pith.science/api/pith-number/XKEB66JJVS5LCBOEW67GYC5PC5/graph.json","events_json":"https://pith.science/api/pith-number/XKEB66JJVS5LCBOEW67GYC5PC5/events.json","paper":"https://pith.science/paper/XKEB66JJ"},"agent_actions":{"view_html":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5","download_json":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5.json","view_paper":"https://pith.science/paper/XKEB66JJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.04793&json=true","fetch_graph":"https://pith.science/api/pith-number/XKEB66JJVS5LCBOEW67GYC5PC5/graph.json","fetch_events":"https://pith.science/api/pith-number/XKEB66JJVS5LCBOEW67GYC5PC5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5/action/storage_attestation","attest_author":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5/action/author_attestation","sign_citation":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5/action/citation_signature","submit_replication":"https://pith.science/pith/XKEB66JJVS5LCBOEW67GYC5PC5/action/replication_record"}},"created_at":"2026-07-05T09:18:24.513729+00:00","updated_at":"2026-07-05T09:18:24.513729+00:00"}