{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5KADX5CLZRC7IZUKB323YH6L53","short_pith_number":"pith:5KADX5CL","schema_version":"1.0","canonical_sha256":"ea803bf44bcc45f4668a0ef5bc1fcbeef679aab45c17e7820ff6e6e20b878152","source":{"kind":"arxiv","id":"2405.14256","version":1},"attestation_state":"computed","paper":{"title":"ZipCache: Accurate and Efficient KV Cache Quantization with Salient Token Identification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bohan Zhuang, Hong Zhou, Jing Liu, Luoming Zhang, Weijia Wu, Yefei He","submitted_at":"2024-05-23T07:37:16Z","abstract_excerpt":"KV cache stores key and value states from previous tokens to avoid re-computation, yet it demands substantial storage space, especially for long sequences. Adaptive KV cache compression seeks to discern the saliency of tokens, preserving vital information while aggressively compressing those of less importance. However, previous methods of this approach exhibit significant performance degradation at high compression ratios due to inaccuracies in identifying salient tokens. In this paper, we present ZipCache, an accurate and efficient KV cache quantization method for LLMs. First, we construct a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14256","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T07:37:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a85f2f6ddf3c4f01db877bbfd865b8bc572e6f65d15a5e10b2c70604f6797279","abstract_canon_sha256":"68b76be837c442239fa2fac40674df18aaaa20ee04b22dfcac05751f717b627a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:12.047025Z","signature_b64":"JYT4CQAxOGUK7v54dawsLXC2SKk/5ojzlV7QcikjRrPxvZPpJc4Ti0SuNklb4o5nAN960KWEY1cO8Yx74JNmCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea803bf44bcc45f4668a0ef5bc1fcbeef679aab45c17e7820ff6e6e20b878152","last_reissued_at":"2026-07-05T08:22:12.046559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:12.046559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ZipCache: Accurate and Efficient KV Cache Quantization with Salient Token Identification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bohan Zhuang, Hong Zhou, Jing Liu, Luoming Zhang, Weijia Wu, Yefei He","submitted_at":"2024-05-23T07:37:16Z","abstract_excerpt":"KV cache stores key and value states from previous tokens to avoid re-computation, yet it demands substantial storage space, especially for long sequences. Adaptive KV cache compression seeks to discern the saliency of tokens, preserving vital information while aggressively compressing those of less importance. However, previous methods of this approach exhibit significant performance degradation at high compression ratios due to inaccuracies in identifying salient tokens. In this paper, we present ZipCache, an accurate and efficient KV cache quantization method for LLMs. First, we construct a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14256","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14256/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14256","created_at":"2026-07-05T08:22:12.046618+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14256v1","created_at":"2026-07-05T08:22:12.046618+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14256","created_at":"2026-07-05T08:22:12.046618+00:00"},{"alias_kind":"pith_short_12","alias_value":"5KADX5CLZRC7","created_at":"2026-07-05T08:22:12.046618+00:00"},{"alias_kind":"pith_short_16","alias_value":"5KADX5CLZRC7IZUK","created_at":"2026-07-05T08:22:12.046618+00:00"},{"alias_kind":"pith_short_8","alias_value":"5KADX5CL","created_at":"2026-07-05T08:22:12.046618+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08302","citing_title":"HACK++: Towards More Effective Head-Aware Key-Value Compression for Efficient Visual Autoregressive Modeling","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01065","citing_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22337","citing_title":"Meta-Soft: Leveraging Composable Meta-Tokens for Context-Preserving KV Cache Compression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25475","citing_title":"IndexMem: Learned KV-Cache Eviction with Latent Memory for Long-Context LLM Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.04390","citing_title":"Through the Stealth Lens: Attention-Aware Defenses Against Poisoning in RAG","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22337","citing_title":"Meta-Soft: Leveraging Composable Meta-Tokens for Context-Preserving KV Cache Compression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18856","citing_title":"SPHERICAL KV: Angle-Domain Attention and Rate-Distortion Retention for Efficient Long-Context Inference","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53","json":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53.json","graph_json":"https://pith.science/api/pith-number/5KADX5CLZRC7IZUKB323YH6L53/graph.json","events_json":"https://pith.science/api/pith-number/5KADX5CLZRC7IZUKB323YH6L53/events.json","paper":"https://pith.science/paper/5KADX5CL"},"agent_actions":{"view_html":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53","download_json":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53.json","view_paper":"https://pith.science/paper/5KADX5CL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14256&json=true","fetch_graph":"https://pith.science/api/pith-number/5KADX5CLZRC7IZUKB323YH6L53/graph.json","fetch_events":"https://pith.science/api/pith-number/5KADX5CLZRC7IZUKB323YH6L53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53/action/storage_attestation","attest_author":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53/action/author_attestation","sign_citation":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53/action/citation_signature","submit_replication":"https://pith.science/pith/5KADX5CLZRC7IZUKB323YH6L53/action/replication_record"}},"created_at":"2026-07-05T08:22:12.046618+00:00","updated_at":"2026-07-05T08:22:12.046618+00:00"}