{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5NYS2JVS4L6KUGMFOZCUERWJG4","short_pith_number":"pith:5NYS2JVS","schema_version":"1.0","canonical_sha256":"eb712d26b2e2fcaa198576454246c93720012c3a0717a5257509fc3bf028b0c5","source":{"kind":"arxiv","id":"2410.15704","version":1},"attestation_state":"computed","paper":{"title":"Residual vector quantization for KV cache compression in large language model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ankur Kumar","submitted_at":"2024-10-21T07:20:41Z","abstract_excerpt":"KV cache compression methods have mainly relied on scalar quantization techniques to reduce the memory requirements during decoding. In this work, we apply residual vector quantization, which has been widely used for high fidelity audio compression, to compress KV cache in large language models (LLM). We adapt the standard recipe with minimal changes to compress the output of any key or value projection matrix in a pretrained LLM: we scale the vector by its standard deviation, divide channels into groups and then quantize each group with the same residual vector quantizer. We learn the codeboo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15704","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-21T07:20:41Z","cross_cats_sorted":[],"title_canon_sha256":"b421cc213e7f488d521affce1aa9527345e0eae0254612b25e582012a09e3618","abstract_canon_sha256":"b642eea39afa5d440ed699c6321a48eb0013714764683af200a0a32d5ab9555e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:23.421114Z","signature_b64":"h/FVZZzsxvRJDLt1W5Vz7Hm9bdWh5LJnYuKly9q8Bz+kM4hJ3TrEEVqKszNoMqoi3Y+O5sFMBmICY5i7srgZCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb712d26b2e2fcaa198576454246c93720012c3a0717a5257509fc3bf028b0c5","last_reissued_at":"2026-07-05T09:23:23.420698Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:23.420698Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Residual vector quantization for KV cache compression in large language model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ankur Kumar","submitted_at":"2024-10-21T07:20:41Z","abstract_excerpt":"KV cache compression methods have mainly relied on scalar quantization techniques to reduce the memory requirements during decoding. In this work, we apply residual vector quantization, which has been widely used for high fidelity audio compression, to compress KV cache in large language models (LLM). We adapt the standard recipe with minimal changes to compress the output of any key or value projection matrix in a pretrained LLM: we scale the vector by its standard deviation, divide channels into groups and then quantize each group with the same residual vector quantizer. We learn the codeboo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15704","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15704","created_at":"2026-07-05T09:23:23.420764+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15704v1","created_at":"2026-07-05T09:23:23.420764+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15704","created_at":"2026-07-05T09:23:23.420764+00:00"},{"alias_kind":"pith_short_12","alias_value":"5NYS2JVS4L6K","created_at":"2026-07-05T09:23:23.420764+00:00"},{"alias_kind":"pith_short_16","alias_value":"5NYS2JVS4L6KUGMF","created_at":"2026-07-05T09:23:23.420764+00:00"},{"alias_kind":"pith_short_8","alias_value":"5NYS2JVS","created_at":"2026-07-05T09:23:23.420764+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06523","citing_title":"DepthWeave-KV: Token-Adaptive Cross-Layer Residual Factorization for Long-Context KV Cache Compression","ref_index":98,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01065","citing_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21335","citing_title":"Sub-Token Routing in LoRA for Adaptation and Query-Aware KV Compression","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4","json":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4.json","graph_json":"https://pith.science/api/pith-number/5NYS2JVS4L6KUGMFOZCUERWJG4/graph.json","events_json":"https://pith.science/api/pith-number/5NYS2JVS4L6KUGMFOZCUERWJG4/events.json","paper":"https://pith.science/paper/5NYS2JVS"},"agent_actions":{"view_html":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4","download_json":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4.json","view_paper":"https://pith.science/paper/5NYS2JVS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15704&json=true","fetch_graph":"https://pith.science/api/pith-number/5NYS2JVS4L6KUGMFOZCUERWJG4/graph.json","fetch_events":"https://pith.science/api/pith-number/5NYS2JVS4L6KUGMFOZCUERWJG4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4/action/storage_attestation","attest_author":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4/action/author_attestation","sign_citation":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4/action/citation_signature","submit_replication":"https://pith.science/pith/5NYS2JVS4L6KUGMFOZCUERWJG4/action/replication_record"}},"created_at":"2026-07-05T09:23:23.420764+00:00","updated_at":"2026-07-05T09:23:23.420764+00:00"}