{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WHEZHNLK65SE4YJB7A5TLUJCHX","short_pith_number":"pith:WHEZHNLK","schema_version":"1.0","canonical_sha256":"b1c993b56af7644e6121f83b35d1223ddcae76207e5f956cc6bff237fcf4b2c9","source":{"kind":"arxiv","id":"2407.21018","version":3},"attestation_state":"computed","paper":{"title":"ThinK: Thinner Key Cache by Query-Driven Pruning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Amrita Saha, Aojun Zhou, Caiming Xiong, Doyen Sahoo, HanZe Dong, Lei Wang, Xudong Lu, Yuhui Xu, Zhanming Jie","submitted_at":"2024-07-30T17:59:08Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized the field of natural language processing, achieving unprecedented performance across a variety of applications. However, their increased computational and memory demands present significant challenges, especially when handling long sequences. This paper focuses on the long-context scenario, addressing the inefficiencies in KV cache memory consumption during inference. Unlike existing approaches that optimize the memory based on the sequence length, we identify substantial redundancy in the channel dimension of the KV cache, as indicated by an un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.21018","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-30T17:59:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5d4a8add7322d4b8cc50a524b7f37d8f5730dfe4460be4338a131f868025e754","abstract_canon_sha256":"2bcd52664b09a2724b20ccc6838d20f01c3653995970a82676f943d9a6d0e21f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:39.070841Z","signature_b64":"+3gaLMRUcKZCBcyVxP+5m1PBxOpnXI3qirUKpGwkFYDsTph0a1J+v6lAGUCj+zJm8YWUVicVhtHbrtaGHWfiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b1c993b56af7644e6121f83b35d1223ddcae76207e5f956cc6bff237fcf4b2c9","last_reissued_at":"2026-07-05T10:20:39.070255Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:39.070255Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ThinK: Thinner Key Cache by Query-Driven Pruning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Amrita Saha, Aojun Zhou, Caiming Xiong, Doyen Sahoo, HanZe Dong, Lei Wang, Xudong Lu, Yuhui Xu, Zhanming Jie","submitted_at":"2024-07-30T17:59:08Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized the field of natural language processing, achieving unprecedented performance across a variety of applications. However, their increased computational and memory demands present significant challenges, especially when handling long sequences. This paper focuses on the long-context scenario, addressing the inefficiencies in KV cache memory consumption during inference. Unlike existing approaches that optimize the memory based on the sequence length, we identify substantial redundancy in the channel dimension of the KV cache, as indicated by an un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.21018","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.21018/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.21018","created_at":"2026-07-05T10:20:39.070323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.21018v3","created_at":"2026-07-05T10:20:39.070323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.21018","created_at":"2026-07-05T10:20:39.070323+00:00"},{"alias_kind":"pith_short_12","alias_value":"WHEZHNLK65SE","created_at":"2026-07-05T10:20:39.070323+00:00"},{"alias_kind":"pith_short_16","alias_value":"WHEZHNLK65SE4YJB","created_at":"2026-07-05T10:20:39.070323+00:00"},{"alias_kind":"pith_short_8","alias_value":"WHEZHNLK","created_at":"2026-07-05T10:20:39.070323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28831","citing_title":"HARD-KV: Head-Adaptive Regularization for Decoding-time KV Compression","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19218","citing_title":"Rotation-Aligned Key Channel Pruning for Efficient Vision-Language Model Inference","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22910","citing_title":"EchoKV: Efficient KV Cache Compression via Similarity-Based Reconstruction","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08317","citing_title":"RDKV: Rate-Distortion Bit Allocation for Joint Eviction and Quantization of the KV Cache","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16864","citing_title":"HieraSparse: Hierarchical Semi-Structured Sparse KV Attention","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16983","citing_title":"Graph-Guided Adaptive Channel Elimination for KV Cache Compression","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX","json":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX.json","graph_json":"https://pith.science/api/pith-number/WHEZHNLK65SE4YJB7A5TLUJCHX/graph.json","events_json":"https://pith.science/api/pith-number/WHEZHNLK65SE4YJB7A5TLUJCHX/events.json","paper":"https://pith.science/paper/WHEZHNLK"},"agent_actions":{"view_html":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX","download_json":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX.json","view_paper":"https://pith.science/paper/WHEZHNLK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.21018&json=true","fetch_graph":"https://pith.science/api/pith-number/WHEZHNLK65SE4YJB7A5TLUJCHX/graph.json","fetch_events":"https://pith.science/api/pith-number/WHEZHNLK65SE4YJB7A5TLUJCHX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX/action/storage_attestation","attest_author":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX/action/author_attestation","sign_citation":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX/action/citation_signature","submit_replication":"https://pith.science/pith/WHEZHNLK65SE4YJB7A5TLUJCHX/action/replication_record"}},"created_at":"2026-07-05T10:20:39.070323+00:00","updated_at":"2026-07-05T10:20:39.070323+00:00"}