{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4IHXDMYOPQG4HAFKLT73AUOGHD","short_pith_number":"pith:4IHXDMYO","schema_version":"1.0","canonical_sha256":"e20f71b30e7c0dc380aa5cffb051c638ce2c91f3c9016300c57494d7d97d7ce5","source":{"kind":"arxiv","id":"2407.08454","version":2},"attestation_state":"computed","paper":{"title":"Model Tells You Where to Merge: Adaptive KV Cache Merging for LLMs on Long-Context Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boxiao Jin, Minjia Zhang, Zheng Wang, Zhongzhi Yu","submitted_at":"2024-07-11T12:50:42Z","abstract_excerpt":"How to efficiently serve Large Language Models (LLMs) has become a pressing issue because of their huge computational cost in their autoregressive generation process. To mitigate computational costs, LLMs often employ the KV Cache technique to improve the generation speed. While improving the computational efficiency, the storage requirements of the KV cache are substantial, particularly in long-context scenarios, leading to significant memory consumption. Existing KV cache eviction methods often degrade the performance of LLMs in long-context scenarios due to the information loss introduced b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08454","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-11T12:50:42Z","cross_cats_sorted":[],"title_canon_sha256":"6950441fc66ab5879d15dc5dc4aa8aec0682769d7cbec27f4472a7e0d92efdd9","abstract_canon_sha256":"45372a57f504094046a438d38285172286f061788d755d9d709d15cb2b9a49f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:27.241967Z","signature_b64":"/HqTEnZHgzNPMGviCmPjtEEvrI03ty3TJbh2LD8Q/dwST7l6U3KmmzIjEP2V7uCiAUp/G8RxmiwneUOT0wQVCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e20f71b30e7c0dc380aa5cffb051c638ce2c91f3c9016300c57494d7d97d7ce5","last_reissued_at":"2026-07-05T08:46:27.241509Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:27.241509Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Tells You Where to Merge: Adaptive KV Cache Merging for LLMs on Long-Context Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boxiao Jin, Minjia Zhang, Zheng Wang, Zhongzhi Yu","submitted_at":"2024-07-11T12:50:42Z","abstract_excerpt":"How to efficiently serve Large Language Models (LLMs) has become a pressing issue because of their huge computational cost in their autoregressive generation process. To mitigate computational costs, LLMs often employ the KV Cache technique to improve the generation speed. While improving the computational efficiency, the storage requirements of the KV cache are substantial, particularly in long-context scenarios, leading to significant memory consumption. Existing KV cache eviction methods often degrade the performance of LLMs in long-context scenarios due to the information loss introduced b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08454","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08454/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08454","created_at":"2026-07-05T08:46:27.241567+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08454v2","created_at":"2026-07-05T08:46:27.241567+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08454","created_at":"2026-07-05T08:46:27.241567+00:00"},{"alias_kind":"pith_short_12","alias_value":"4IHXDMYOPQG4","created_at":"2026-07-05T08:46:27.241567+00:00"},{"alias_kind":"pith_short_16","alias_value":"4IHXDMYOPQG4HAFK","created_at":"2026-07-05T08:46:27.241567+00:00"},{"alias_kind":"pith_short_8","alias_value":"4IHXDMYO","created_at":"2026-07-05T08:46:27.241567+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.06519","citing_title":"FreqDepthKV: Frequency-Guided Depth Sharing for Robust KV Cache Compression in Long-Context LLM Inference","ref_index":110,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06827","citing_title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06523","citing_title":"DepthWeave-KV: Token-Adaptive Cross-Layer Residual Factorization for Long-Context KV Cache Compression","ref_index":126,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05698","citing_title":"Rethinking LoRA Memory Through the Lens of KV Cache Compression","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03979","citing_title":"Language Models Need Sleep: Learning to Self-Modify and Consolidate Memories","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02553","citing_title":"LongLive-RAG: A General Retrieval-Augmented Framework for Long Video Generation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01563","citing_title":"MomentKV: Closing the Directional Gap in KV Cache Eviction for Long-Context Inference","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31105","citing_title":"GRKV: Global Regression for Training-Free KV Cache Compression in Long-Context LLMs","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09508","citing_title":"From Rigid to Dynamic: Entropy-Guided Adaptive Inference for Long-Context LLMs","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05772","citing_title":"Sparse Attention Remapping with Clustering for Efficient LLM Decoding on PIM","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21433","citing_title":"ReasonCache: Accelerating Large Reasoning Model Serving through KV Cache Sharing","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20284","citing_title":"STAC: Plug-and-Play Spatio-Temporal Aware Cache Compression for Streaming 3D Reconstruction","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25642","citing_title":"Prefill-Time Intervention for Mitigating Hallucination in Large Vision-Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17935","citing_title":"How Much Cache Does Reasoning Need? Depth-Cache Tradeoffs in KV-Compressed Transformers","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD","json":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD.json","graph_json":"https://pith.science/api/pith-number/4IHXDMYOPQG4HAFKLT73AUOGHD/graph.json","events_json":"https://pith.science/api/pith-number/4IHXDMYOPQG4HAFKLT73AUOGHD/events.json","paper":"https://pith.science/paper/4IHXDMYO"},"agent_actions":{"view_html":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD","download_json":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD.json","view_paper":"https://pith.science/paper/4IHXDMYO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08454&json=true","fetch_graph":"https://pith.science/api/pith-number/4IHXDMYOPQG4HAFKLT73AUOGHD/graph.json","fetch_events":"https://pith.science/api/pith-number/4IHXDMYOPQG4HAFKLT73AUOGHD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD/action/storage_attestation","attest_author":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD/action/author_attestation","sign_citation":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD/action/citation_signature","submit_replication":"https://pith.science/pith/4IHXDMYOPQG4HAFKLT73AUOGHD/action/replication_record"}},"created_at":"2026-07-05T08:46:27.241567+00:00","updated_at":"2026-07-05T08:46:27.241567+00:00"}