{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZYCLVPGIEE3I5Z47MP5D6MXFNO","short_pith_number":"pith:ZYCLVPGI","schema_version":"1.0","canonical_sha256":"ce04babcc821368ee79f63fa3f32e56bb332fff2e53882d539c86aa6eab4e16c","source":{"kind":"arxiv","id":"2405.14366","version":2},"attestation_state":"computed","paper":{"title":"MiniCache: KV Cache Compression in Depth Dimension for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Akide Liu, Bohan Zhuang, Gholamreza Haffari, Jing Liu, Yefei He, Zizheng Pan","submitted_at":"2024-05-23T09:43:52Z","abstract_excerpt":"A critical approach for efficiently deploying computationally demanding large language models (LLMs) is Key-Value (KV) caching. The KV cache stores key-value states of previously generated tokens, significantly reducing the need for repetitive computations and thereby lowering latency in autoregressive generation. However, the size of the KV cache grows linearly with sequence length, posing challenges for applications requiring long context input and extensive sequence generation. In this paper, we present a simple yet effective approach, called MiniCache, to compress the KV cache across layer"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14366","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T09:43:52Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"44d2bd79a6f1e720a7f18e530c969e3a6adccf725b3b74b4aea5029d8fe33b23","abstract_canon_sha256":"414f1093f7e476a2274c29b924625394fc0486bcd2eb1435fc5ef727d059a9b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:10.032636Z","signature_b64":"WhovnsPfV+tFwb/i7Hg3h6aXTR7dJEAaD6OtmtuoIeQ+3T46tMFD1up7u5pee/rcEJQrdBRT1+JFdtV1xCeMBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce04babcc821368ee79f63fa3f32e56bb332fff2e53882d539c86aa6eab4e16c","last_reissued_at":"2026-07-05T09:04:10.032158Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:10.032158Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MiniCache: KV Cache Compression in Depth Dimension for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Akide Liu, Bohan Zhuang, Gholamreza Haffari, Jing Liu, Yefei He, Zizheng Pan","submitted_at":"2024-05-23T09:43:52Z","abstract_excerpt":"A critical approach for efficiently deploying computationally demanding large language models (LLMs) is Key-Value (KV) caching. The KV cache stores key-value states of previously generated tokens, significantly reducing the need for repetitive computations and thereby lowering latency in autoregressive generation. However, the size of the KV cache grows linearly with sequence length, posing challenges for applications requiring long context input and extensive sequence generation. In this paper, we present a simple yet effective approach, called MiniCache, to compress the KV cache across layer"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14366","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14366/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14366","created_at":"2026-07-05T09:04:10.032216+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14366v2","created_at":"2026-07-05T09:04:10.032216+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14366","created_at":"2026-07-05T09:04:10.032216+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYCLVPGIEE3I","created_at":"2026-07-05T09:04:10.032216+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYCLVPGIEE3I5Z47","created_at":"2026-07-05T09:04:10.032216+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYCLVPGI","created_at":"2026-07-05T09:04:10.032216+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.06519","citing_title":"FreqDepthKV: Frequency-Guided Depth Sharing for Robust KV Cache Compression in Long-Context LLM Inference","ref_index":89,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":71,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06523","citing_title":"DepthWeave-KV: Token-Adaptive Cross-Layer Residual Factorization for Long-Context KV Cache Compression","ref_index":103,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24033","citing_title":"RoPE-Aware Bit Allocation for KV-Cache Quantization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08302","citing_title":"HACK++: Towards More Effective Head-Aware Key-Value Compression for Efficient Visual Autoregressive Modeling","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01941","citing_title":"Semantic Integrity Matters: Benchmarking and Preserving High-Density Reasoning in KV Cache Compression","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2503.19950","citing_title":"LogQuant: Log-Distributed 2-Bit Quantization of KV Cache with Superior Accuracy Preservation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22575","citing_title":"S2O: Early Stopping for Sparse Attention via Online Permutation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO","json":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO.json","graph_json":"https://pith.science/api/pith-number/ZYCLVPGIEE3I5Z47MP5D6MXFNO/graph.json","events_json":"https://pith.science/api/pith-number/ZYCLVPGIEE3I5Z47MP5D6MXFNO/events.json","paper":"https://pith.science/paper/ZYCLVPGI"},"agent_actions":{"view_html":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO","download_json":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO.json","view_paper":"https://pith.science/paper/ZYCLVPGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14366&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYCLVPGIEE3I5Z47MP5D6MXFNO/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYCLVPGIEE3I5Z47MP5D6MXFNO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO/action/storage_attestation","attest_author":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO/action/author_attestation","sign_citation":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO/action/citation_signature","submit_replication":"https://pith.science/pith/ZYCLVPGIEE3I5Z47MP5D6MXFNO/action/replication_record"}},"created_at":"2026-07-05T09:04:10.032216+00:00","updated_at":"2026-07-05T09:04:10.032216+00:00"}