{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ABSH3VJSHK2ZSCINU2ZODWEKLT","short_pith_number":"pith:ABSH3VJS","schema_version":"1.0","canonical_sha256":"00647dd5323ab599090da6b2e1d88a5cec04cb3bf8f8d81a682c5e1a674fe82a","source":{"kind":"arxiv","id":"2501.19392","version":4},"attestation_state":"computed","paper":{"title":"Cache Me If You Must: Adaptive Key-Value Quantization for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alina Shutova, Dan Alistarh, Denis Kuznedelev, Denis Mazur, Ivan Ermakov, Nikita Surkov, Vage Egiazarian, Vladimir Malinovskii","submitted_at":"2025-01-31T18:47:42Z","abstract_excerpt":"Efficient real-world deployments of large language models (LLMs) rely on Key-Value (KV) caching for processing and generating long outputs, reducing the need for repetitive computation. For large contexts, Key-Value caches can take up tens of gigabytes of device memory, as they store vector representations for each token and layer. Recent work has shown that the cached vectors can be compressed through quantization, pruning or merging, but these techniques often compromise quality towards higher compression rates. In this work, we aim to improve Key & Value compression by exploiting two observ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.19392","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-31T18:47:42Z","cross_cats_sorted":[],"title_canon_sha256":"965422ea026d169eb6768c0b621052f73beebb2a2a1c9c5a6ca830adf3429141","abstract_canon_sha256":"18075049eee8cc56855753f5c5fbc763eccb54eb0327eb86099ae14ee9876003"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:52.269006Z","signature_b64":"/j+bjVfKcJ7BlxySHaT+JN07Y12XjDJZUCrLYUdqsS1caQboY+3/9fBWukDoinPfA3D/n/FKGrR4n/VfSrNjDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00647dd5323ab599090da6b2e1d88a5cec04cb3bf8f8d81a682c5e1a674fe82a","last_reissued_at":"2026-07-05T10:21:52.268493Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:52.268493Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cache Me If You Must: Adaptive Key-Value Quantization for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alina Shutova, Dan Alistarh, Denis Kuznedelev, Denis Mazur, Ivan Ermakov, Nikita Surkov, Vage Egiazarian, Vladimir Malinovskii","submitted_at":"2025-01-31T18:47:42Z","abstract_excerpt":"Efficient real-world deployments of large language models (LLMs) rely on Key-Value (KV) caching for processing and generating long outputs, reducing the need for repetitive computation. For large contexts, Key-Value caches can take up tens of gigabytes of device memory, as they store vector representations for each token and layer. Recent work has shown that the cached vectors can be compressed through quantization, pruning or merging, but these techniques often compromise quality towards higher compression rates. In this work, we aim to improve Key & Value compression by exploiting two observ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.19392","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.19392/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.19392","created_at":"2026-07-05T10:21:52.268554+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.19392v4","created_at":"2026-07-05T10:21:52.268554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.19392","created_at":"2026-07-05T10:21:52.268554+00:00"},{"alias_kind":"pith_short_12","alias_value":"ABSH3VJSHK2Z","created_at":"2026-07-05T10:21:52.268554+00:00"},{"alias_kind":"pith_short_16","alias_value":"ABSH3VJSHK2ZSCIN","created_at":"2026-07-05T10:21:52.268554+00:00"},{"alias_kind":"pith_short_8","alias_value":"ABSH3VJS","created_at":"2026-07-05T10:21:52.268554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23258","citing_title":"A Simple Plug-in for Improving Eviction-Based KV Cache Compression","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08426","citing_title":"KV Cache Offloading for Context-Intensive Tasks","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08426","citing_title":"KV Cache Offloading for Context-Intensive Tasks","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08317","citing_title":"RDKV: Rate-Distortion Bit Allocation for Joint Eviction and Quantization of the KV Cache","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08426","citing_title":"KV Cache Offloading for Context-Intensive Tasks","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08426","citing_title":"KV Cache Offloading for Context-Intensive Tasks","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT","json":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT.json","graph_json":"https://pith.science/api/pith-number/ABSH3VJSHK2ZSCINU2ZODWEKLT/graph.json","events_json":"https://pith.science/api/pith-number/ABSH3VJSHK2ZSCINU2ZODWEKLT/events.json","paper":"https://pith.science/paper/ABSH3VJS"},"agent_actions":{"view_html":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT","download_json":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT.json","view_paper":"https://pith.science/paper/ABSH3VJS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.19392&json=true","fetch_graph":"https://pith.science/api/pith-number/ABSH3VJSHK2ZSCINU2ZODWEKLT/graph.json","fetch_events":"https://pith.science/api/pith-number/ABSH3VJSHK2ZSCINU2ZODWEKLT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT/action/storage_attestation","attest_author":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT/action/author_attestation","sign_citation":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT/action/citation_signature","submit_replication":"https://pith.science/pith/ABSH3VJSHK2ZSCINU2ZODWEKLT/action/replication_record"}},"created_at":"2026-07-05T10:21:52.268554+00:00","updated_at":"2026-07-05T10:21:52.268554+00:00"}