{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZKGBG2ZCLP4VQ534CSTM3KNPS5","short_pith_number":"pith:ZKGBG2ZC","schema_version":"1.0","canonical_sha256":"ca8c136b225bf958777c14a6cda9af977045e9283a19d387a00110255b24f809","source":{"kind":"arxiv","id":"2401.18079","version":6},"attestation_state":"computed","paper":{"title":"KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amir Gholami, Coleman Hooper, Hiva Mohammadzadeh, Kurt Keutzer, Michael W. Mahoney, Sehoon Kim, Yakun Sophia Shao","submitted_at":"2024-01-31T18:58:14Z","abstract_excerpt":"LLMs are seeing growing use for applications which require large context windows, and with these large context windows KV cache activations surface as the dominant contributor to memory consumption during inference. Quantization is a promising approach for compressing KV cache activations; however, existing solutions fail to represent activations accurately in sub-4-bit precision. Our work, KVQuant, facilitates low precision KV cache quantization by incorporating several novel methods: (i) Per-Channel Key Quantization, where we adjust the dimension along which we quantize the Key activations t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.18079","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-01-31T18:58:14Z","cross_cats_sorted":[],"title_canon_sha256":"8093b554e94f847fdf86355a6fc1785b19344f5fed33a8a634eefd2b4794cd4e","abstract_canon_sha256":"777a4a4bc787e4fb21f6caaab70dbbdf32806629d57e4442de0190a263ce18b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:30.840105Z","signature_b64":"HiexR79w3UONsmauWodfAZaVpwOhwjG/0+JodsUV9ZacgdtbwnqlsEzdEmBwvXDHgjibg8kMSlgkq19W/ynIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca8c136b225bf958777c14a6cda9af977045e9283a19d387a00110255b24f809","last_reissued_at":"2026-07-05T11:11:30.838685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:30.838685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amir Gholami, Coleman Hooper, Hiva Mohammadzadeh, Kurt Keutzer, Michael W. Mahoney, Sehoon Kim, Yakun Sophia Shao","submitted_at":"2024-01-31T18:58:14Z","abstract_excerpt":"LLMs are seeing growing use for applications which require large context windows, and with these large context windows KV cache activations surface as the dominant contributor to memory consumption during inference. Quantization is a promising approach for compressing KV cache activations; however, existing solutions fail to represent activations accurately in sub-4-bit precision. Our work, KVQuant, facilitates low precision KV cache quantization by incorporating several novel methods: (i) Per-Channel Key Quantization, where we adjust the dimension along which we quantize the Key activations t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.18079","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.18079/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.18079","created_at":"2026-07-05T11:11:30.838744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.18079v6","created_at":"2026-07-05T11:11:30.838744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.18079","created_at":"2026-07-05T11:11:30.838744+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZKGBG2ZCLP4V","created_at":"2026-07-05T11:11:30.838744+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZKGBG2ZCLP4VQ534","created_at":"2026-07-05T11:11:30.838744+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZKGBG2ZC","created_at":"2026-07-05T11:11:30.838744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07144","citing_title":"Fractal KV-Cache Archives: Lossless Symbolic Storage with In-Place Retrieval for Long-Context LLM Inference","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08382","citing_title":"STAR-KV: Low-Rank KV Cache Compression via Soft Thresholding for Adaptive Rank Control","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01065","citing_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02780","citing_title":"Do Value Vectors in Deep Layers Need Context from the Residual Stream?","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09864","citing_title":"Alignment Collapse Under KV Cache Quantization: Diagnosis and Mitigation","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25203","citing_title":"Influence-Inspired Spectral Rotations for Extreme Low-Bit LLM Quantization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09508","citing_title":"From Rigid to Dynamic: Entropy-Guided Adaptive Inference for Long-Context LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23200","citing_title":"Adaptive Mass-Segmented KV Compression for Long-Context Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17396","citing_title":"EpiCache: Episodic KV Cache Management for Long-Term Conversation on Resource-Constrained Environments","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20868","citing_title":"Runtime-Certified Bounded-Error Quantized Attention","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18856","citing_title":"SPHERICAL KV: Angle-Domain Attention and Rate-Distortion Retention for Efficient Long-Context Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2407.08608","citing_title":"FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2312.05821","citing_title":"ASVD: Activation-aware Singular Value Decomposition for Compressing Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18053","citing_title":"Protection Is (Nearly) All You Need: Structural Protection Dominates Scoring in Globally Capped KV Eviction","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19874","citing_title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03562","citing_title":"HeadQ: Model-Visible Distortion and Score-Space Correction for KV-Cache Quantization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2402.02750","citing_title":"KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2312.07104","citing_title":"SGLang: Efficient Execution of Structured Language Model Programs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08966","citing_title":"VORT: Adaptive Power-Law Memory for NLP Transformers","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24971","citing_title":"PolyKV: A Shared Asymmetrically-Compressed KV Cache Pool for Multi-Agent LLM Inference","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02262","citing_title":"WindowQuant: Mixed-Precision KV Cache Quantization based on Window-Level Similarity for VLMs Inference Optimization","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15356","citing_title":"Sequential KV Cache Compression via Probabilistic Language Tries: Beyond the Per-Vector Shannon Limit","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2405.04434","citing_title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5","json":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5.json","graph_json":"https://pith.science/api/pith-number/ZKGBG2ZCLP4VQ534CSTM3KNPS5/graph.json","events_json":"https://pith.science/api/pith-number/ZKGBG2ZCLP4VQ534CSTM3KNPS5/events.json","paper":"https://pith.science/paper/ZKGBG2ZC"},"agent_actions":{"view_html":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5","download_json":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5.json","view_paper":"https://pith.science/paper/ZKGBG2ZC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.18079&json=true","fetch_graph":"https://pith.science/api/pith-number/ZKGBG2ZCLP4VQ534CSTM3KNPS5/graph.json","fetch_events":"https://pith.science/api/pith-number/ZKGBG2ZCLP4VQ534CSTM3KNPS5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5/action/storage_attestation","attest_author":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5/action/author_attestation","sign_citation":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5/action/citation_signature","submit_replication":"https://pith.science/pith/ZKGBG2ZCLP4VQ534CSTM3KNPS5/action/replication_record"}},"created_at":"2026-07-05T11:11:30.838744+00:00","updated_at":"2026-07-05T11:11:30.838744+00:00"}