{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3SCIASRYSX2QD6DX657SR4GBVP","short_pith_number":"pith:3SCIASRY","schema_version":"1.0","canonical_sha256":"dc84804a3895f501f877f77f28f0c1abf2d19fe7ef7b8c1fde77fa30fc49b856","source":{"kind":"arxiv","id":"2412.14838","version":4},"attestation_state":"computed","paper":{"title":"DynamicKV: Task-Aware Adaptive KV Cache Compression for Long Context LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiaxian Guo, Liang Ding, Li Shen, Minyan Zeng, Min Zhang, Wenbin Wang, Xiabin Zhou, Xuebo Liu","submitted_at":"2024-12-19T13:28:42Z","abstract_excerpt":"Efficient KV cache management in LLMs is crucial for long-context tasks like RAG and summarization. Existing KV cache compression methods enforce a fixed pattern, neglecting task-specific characteristics and reducing the retention of essential information. However, we observe distinct activation patterns across layers in various tasks, highlighting the need for adaptive strategies tailored to each task's unique demands. Based on this insight, we propose DynamicKV, a method that dynamically optimizes token retention by adjusting the number of tokens retained at each layer to adapt to the specif"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.14838","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-19T13:28:42Z","cross_cats_sorted":[],"title_canon_sha256":"1921d4b483aa1c7b7db3a443df6da0ac24de4fd29b7a2280ac2cbf2f801566b7","abstract_canon_sha256":"45813f1a99a26700d288abb45a6fb32b2b94612ba4bb23e58e8e8e91f7fea6f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:03.363564Z","signature_b64":"sFyoBehZwzQkCpz3TJSHYJyCSjMuPdcPkj0t4skTO1iYfIHvoCRSMH1Ql1YYr4TYwjVj91h8TnoIAvLUcZMxAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc84804a3895f501f877f77f28f0c1abf2d19fe7ef7b8c1fde77fa30fc49b856","last_reissued_at":"2026-07-05T11:10:03.363055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:03.363055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DynamicKV: Task-Aware Adaptive KV Cache Compression for Long Context LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiaxian Guo, Liang Ding, Li Shen, Minyan Zeng, Min Zhang, Wenbin Wang, Xiabin Zhou, Xuebo Liu","submitted_at":"2024-12-19T13:28:42Z","abstract_excerpt":"Efficient KV cache management in LLMs is crucial for long-context tasks like RAG and summarization. Existing KV cache compression methods enforce a fixed pattern, neglecting task-specific characteristics and reducing the retention of essential information. However, we observe distinct activation patterns across layers in various tasks, highlighting the need for adaptive strategies tailored to each task's unique demands. Based on this insight, we propose DynamicKV, a method that dynamically optimizes token retention by adjusting the number of tokens retained at each layer to adapt to the specif"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.14838","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.14838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.14838","created_at":"2026-07-05T11:10:03.363114+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.14838v4","created_at":"2026-07-05T11:10:03.363114+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.14838","created_at":"2026-07-05T11:10:03.363114+00:00"},{"alias_kind":"pith_short_12","alias_value":"3SCIASRYSX2Q","created_at":"2026-07-05T11:10:03.363114+00:00"},{"alias_kind":"pith_short_16","alias_value":"3SCIASRYSX2QD6DX","created_at":"2026-07-05T11:10:03.363114+00:00"},{"alias_kind":"pith_short_8","alias_value":"3SCIASRY","created_at":"2026-07-05T11:10:03.363114+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20246","citing_title":"Finetuning Vision-Language-Action Models Requires Fewer Layers Than You Think","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17872","citing_title":"AnchorKV: Safety-Aware KV Cache Compression via Soft Penalty with a Refusal Anchor","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11164","citing_title":"ReasonAlloc: Hierarchical Decoding-Time KV Cache Budget Allocation for Reasoning Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08635","citing_title":"SpectrumKV: Per-Token Mixed-Precision KV Cache Transfer for Prefill-Decode Disaggregated LLM Serving","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09916","citing_title":"IntentKV: Cross-Turn Intent-Aware KV Cache Pruning for Agent Inference","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05868","citing_title":"YouZhi: Towards High-Concurrency Financial LLMs via Adaptive GQA-to-MLA Transition","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08686","citing_title":"CompilerKV: Risk-Adaptive KV Compression via Offline Experience Compilation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04592","citing_title":"From Static Inference to Dynamic Interaction: A Survey of Streaming Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08840","citing_title":"ReST-KV: Robust KV Cache Eviction with Layer-wise Output Reconstruction and Spatial-Temporal Smoothing","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":211,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04722","citing_title":"Don't Waste Bits! Adaptive KV-Cache Quantization for Lightweight On-Device LLMs","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15583","citing_title":"SAGE: Selective Attention-Guided Extraction for Token-Efficient Document Indexing","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16864","citing_title":"HieraSparse: Hierarchical Semi-Structured Sparse KV Attention","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP","json":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP.json","graph_json":"https://pith.science/api/pith-number/3SCIASRYSX2QD6DX657SR4GBVP/graph.json","events_json":"https://pith.science/api/pith-number/3SCIASRYSX2QD6DX657SR4GBVP/events.json","paper":"https://pith.science/paper/3SCIASRY"},"agent_actions":{"view_html":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP","download_json":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP.json","view_paper":"https://pith.science/paper/3SCIASRY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.14838&json=true","fetch_graph":"https://pith.science/api/pith-number/3SCIASRYSX2QD6DX657SR4GBVP/graph.json","fetch_events":"https://pith.science/api/pith-number/3SCIASRYSX2QD6DX657SR4GBVP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP/action/storage_attestation","attest_author":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP/action/author_attestation","sign_citation":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP/action/citation_signature","submit_replication":"https://pith.science/pith/3SCIASRYSX2QD6DX657SR4GBVP/action/replication_record"}},"created_at":"2026-07-05T11:10:03.363114+00:00","updated_at":"2026-07-05T11:10:03.363114+00:00"}