{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KDVJUZN7CXZP42TII3EZB3DY52","short_pith_number":"pith:KDVJUZN7","schema_version":"1.0","canonical_sha256":"50ea9a65bf15f2fe6a6846c990ec78ee887137eafb528b7e2cedc02e89cf6661","source":{"kind":"arxiv","id":"2407.18003","version":4},"attestation_state":"computed","paper":{"title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hai Zhao, Hongyi Zhang, Luohe Shi, Yao Yao, Zuchao Li","submitted_at":"2024-07-25T12:56:22Z","abstract_excerpt":"Large Language Models (LLMs), epitomized by ChatGPT's release in late 2022, have revolutionized various industries with their advanced language comprehension. However, their efficiency is challenged by the Transformer architecture's struggle with handling long texts. KV Cache has emerged as a pivotal solution to this issue, converting the time complexity of token generation from quadratic to linear, albeit with increased GPU memory overhead proportional to conversation length. With the development of the LLM community and academia, various KV Cache compression methods have been proposed. In th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.18003","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-25T12:56:22Z","cross_cats_sorted":[],"title_canon_sha256":"545fd3feabd0018ad6047b13d59d8fbcfb577eb30a8f4411063673b153895397","abstract_canon_sha256":"a8e575d8cc125f050562a57edea38dd7fe564af0dbc85af8e893fdedb734ba6c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:37:55.735960Z","signature_b64":"pXwXrt0wbdRagOOfZRkC5vRDZJ8Srqh2o5eMONRZM0e+1F6KVcOkXVmc+13C+xxPatd2qp3yg0hfkRWakje3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50ea9a65bf15f2fe6a6846c990ec78ee887137eafb528b7e2cedc02e89cf6661","last_reissued_at":"2026-07-05T09:37:55.735437Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:37:55.735437Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hai Zhao, Hongyi Zhang, Luohe Shi, Yao Yao, Zuchao Li","submitted_at":"2024-07-25T12:56:22Z","abstract_excerpt":"Large Language Models (LLMs), epitomized by ChatGPT's release in late 2022, have revolutionized various industries with their advanced language comprehension. However, their efficiency is challenged by the Transformer architecture's struggle with handling long texts. KV Cache has emerged as a pivotal solution to this issue, converting the time complexity of token generation from quadratic to linear, albeit with increased GPU memory overhead proportional to conversation length. With the development of the LLM community and academia, various KV Cache compression methods have been proposed. In th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.18003","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.18003/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.18003","created_at":"2026-07-05T09:37:55.735491+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.18003v4","created_at":"2026-07-05T09:37:55.735491+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.18003","created_at":"2026-07-05T09:37:55.735491+00:00"},{"alias_kind":"pith_short_12","alias_value":"KDVJUZN7CXZP","created_at":"2026-07-05T09:37:55.735491+00:00"},{"alias_kind":"pith_short_16","alias_value":"KDVJUZN7CXZP42TI","created_at":"2026-07-05T09:37:55.735491+00:00"},{"alias_kind":"pith_short_8","alias_value":"KDVJUZN7","created_at":"2026-07-05T09:37:55.735491+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00620","citing_title":"FlowNar: Scalable Streaming Narration for Long-Form Videos","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13846","citing_title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2502.20295","citing_title":"Judge a Book by its Cover: Investigating Multi-Modal LLMs for Multi-Page Handwritten Document Transcription","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21623","citing_title":"OjaKV: Context-Aware Online Low-Rank KV Cache Compression","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2602.10718","citing_title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22910","citing_title":"EchoKV: Efficient KV Cache Compression via Similarity-Based Reconstruction","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11516","citing_title":"Agents Should Replace Narrow Predictive AI as the Orchestrator in 6G AI-RAN","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04901","citing_title":"On the (In-)Security of the Shuffling Defense in the Transformer Secure Inference","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04921","citing_title":"TriAttention: Efficient Long Reasoning with Trigonometric KV Compression","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04500","citing_title":"Saliency-R1: Enforcing Interpretable and Faithful Vision-language Reasoning via Saliency-map Alignment Reward","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17935","citing_title":"How Much Cache Does Reasoning Need? Depth-Cache Tradeoffs in KV-Compressed Transformers","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25975","citing_title":"Rethinking KV Cache Eviction via a Unified Information-Theoretic Objective","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52","json":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52.json","graph_json":"https://pith.science/api/pith-number/KDVJUZN7CXZP42TII3EZB3DY52/graph.json","events_json":"https://pith.science/api/pith-number/KDVJUZN7CXZP42TII3EZB3DY52/events.json","paper":"https://pith.science/paper/KDVJUZN7"},"agent_actions":{"view_html":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52","download_json":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52.json","view_paper":"https://pith.science/paper/KDVJUZN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.18003&json=true","fetch_graph":"https://pith.science/api/pith-number/KDVJUZN7CXZP42TII3EZB3DY52/graph.json","fetch_events":"https://pith.science/api/pith-number/KDVJUZN7CXZP42TII3EZB3DY52/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52/action/storage_attestation","attest_author":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52/action/author_attestation","sign_citation":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52/action/citation_signature","submit_replication":"https://pith.science/pith/KDVJUZN7CXZP42TII3EZB3DY52/action/replication_record"}},"created_at":"2026-07-05T09:37:55.735491+00:00","updated_at":"2026-07-05T09:37:55.735491+00:00"}