{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5TXPIUIPS7JPFLUMIJVH5U7XAU","short_pith_number":"pith:5TXPIUIP","schema_version":"1.0","canonical_sha256":"eceef4510f97d2f2ae8c426a7ed3f70537ac8abda41c5e401f3d5747f79c7788","source":{"kind":"arxiv","id":"2508.02401","version":1},"attestation_state":"computed","paper":{"title":"CompressKV: Semantic Retrieval Heads Know What Tokens are Not Important Before Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bing Li, Grace Li Zhang, Jingcun Wang, Olga Kondrateva, Xiaolin Lin, Yiyu Shi","submitted_at":"2025-08-04T13:26:16Z","abstract_excerpt":"Recent advances in large language models (LLMs) have significantly boosted long-context processing. However, the increasing key-value (KV) cache size poses critical challenges to memory and execution efficiency. Most KV cache compression methods rely on heuristic token eviction using all attention heads in Grouped Query Attention (GQA)-based LLMs. This method ignores the different functionalities of attention heads, leading to the eviction of critical tokens and thus degrades the performance of LLMs.\n  To address the issue above, instead of using all the attention heads in GQA-based LLMs to de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.02401","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-04T13:26:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ae8791caaf8f95f5d3d4cf605dbe759d134e74a72594aa0a75203133d789dc35","abstract_canon_sha256":"df0a68b5b7327d643f36814c5391a7bf336ea7f12736e190632fe3033995ac38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:13.788029Z","signature_b64":"ohdW+VbNZjHYUFPdN4a9HdZuLRL/6GpJ13oJHNcO2mwwXgK8Vm6Bj5XyZP6HNtZF5Fsy02ntZ3r+obq0IK3QAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eceef4510f97d2f2ae8c426a7ed3f70537ac8abda41c5e401f3d5747f79c7788","last_reissued_at":"2026-07-05T11:48:13.787475Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:13.787475Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CompressKV: Semantic Retrieval Heads Know What Tokens are Not Important Before Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bing Li, Grace Li Zhang, Jingcun Wang, Olga Kondrateva, Xiaolin Lin, Yiyu Shi","submitted_at":"2025-08-04T13:26:16Z","abstract_excerpt":"Recent advances in large language models (LLMs) have significantly boosted long-context processing. However, the increasing key-value (KV) cache size poses critical challenges to memory and execution efficiency. Most KV cache compression methods rely on heuristic token eviction using all attention heads in Grouped Query Attention (GQA)-based LLMs. This method ignores the different functionalities of attention heads, leading to the eviction of critical tokens and thus degrades the performance of LLMs.\n  To address the issue above, instead of using all the attention heads in GQA-based LLMs to de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.02401","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.02401/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.02401","created_at":"2026-07-05T11:48:13.787534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.02401v1","created_at":"2026-07-05T11:48:13.787534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.02401","created_at":"2026-07-05T11:48:13.787534+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TXPIUIPS7JP","created_at":"2026-07-05T11:48:13.787534+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TXPIUIPS7JPFLUM","created_at":"2026-07-05T11:48:13.787534+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TXPIUIP","created_at":"2026-07-05T11:48:13.787534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24957","citing_title":"Dustin: Draft-Augmented Sparse Verification for Efficient Long-Context Generation with Speculative Decoding","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06256","citing_title":"RedKnot: Efficient Long-Context LLM Serving with Head-Aware KV Reuse and SegPagedAttention","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01002","citing_title":"Logit-Contribution Scoring Identifies Non-Literal Retrieval Heads","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06256","citing_title":"RedKnot: Efficient Long-Context LLM Serving with Head-Aware KV Reuse and SegPagedAttention","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU","json":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU.json","graph_json":"https://pith.science/api/pith-number/5TXPIUIPS7JPFLUMIJVH5U7XAU/graph.json","events_json":"https://pith.science/api/pith-number/5TXPIUIPS7JPFLUMIJVH5U7XAU/events.json","paper":"https://pith.science/paper/5TXPIUIP"},"agent_actions":{"view_html":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU","download_json":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU.json","view_paper":"https://pith.science/paper/5TXPIUIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.02401&json=true","fetch_graph":"https://pith.science/api/pith-number/5TXPIUIPS7JPFLUMIJVH5U7XAU/graph.json","fetch_events":"https://pith.science/api/pith-number/5TXPIUIPS7JPFLUMIJVH5U7XAU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU/action/storage_attestation","attest_author":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU/action/author_attestation","sign_citation":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU/action/citation_signature","submit_replication":"https://pith.science/pith/5TXPIUIPS7JPFLUMIJVH5U7XAU/action/replication_record"}},"created_at":"2026-07-05T11:48:13.787534+00:00","updated_at":"2026-07-05T11:48:13.787534+00:00"}