{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OSQTHM6WV4MTWQFHDA6FMK76Q3","short_pith_number":"pith:OSQTHM6W","schema_version":"1.0","canonical_sha256":"74a133b3d6af193b40a7183c562bfe86de1568f233973348bbe159ae926c7f77","source":{"kind":"arxiv","id":"2412.03213","version":2},"attestation_state":"computed","paper":{"title":"ClusterKV: Manipulating LLM KV Cache in Semantic Space for Recallable Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.PF"],"primary_cat":"cs.LG","authors_text":"Chengwei Li, Chenqi Zhang, Guangda Liu, Jieru Zhao, Minyi Guo","submitted_at":"2024-12-04T10:58:27Z","abstract_excerpt":"Large Language Models (LLMs) have been widely deployed in a variety of applications, and the context length is rapidly increasing to handle tasks such as long-document QA and complex logical reasoning. However, long context poses significant challenges for inference efficiency, including high memory costs of key-value (KV) cache and increased latency due to extensive memory accesses. Recent works have proposed compressing KV cache to approximate computation, but these methods either evict tokens permanently, never recalling them for later inference, or recall previous tokens at the granularity"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.03213","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-04T10:58:27Z","cross_cats_sorted":["cs.AI","cs.PF"],"title_canon_sha256":"1e900003cc27815d965f5dc68134a0efe4aecdef37b8b80de296c26240bf5d80","abstract_canon_sha256":"97ad5786b8da64b8b0fb5c3fceed0d7254fbbce828ba6419344fac8968a6ece7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:38.826373Z","signature_b64":"+AvqbvMUAxdaWUSonbCJFZhyenMP9dXZbNPEzRS98Q5qE1VdcupjNsxZxOnMekWPvJ6zSY96ej50TEokpQWuBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74a133b3d6af193b40a7183c562bfe86de1568f233973348bbe159ae926c7f77","last_reissued_at":"2026-07-05T11:21:38.825854Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:38.825854Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ClusterKV: Manipulating LLM KV Cache in Semantic Space for Recallable Compression","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.PF"],"primary_cat":"cs.LG","authors_text":"Chengwei Li, Chenqi Zhang, Guangda Liu, Jieru Zhao, Minyi Guo","submitted_at":"2024-12-04T10:58:27Z","abstract_excerpt":"Large Language Models (LLMs) have been widely deployed in a variety of applications, and the context length is rapidly increasing to handle tasks such as long-document QA and complex logical reasoning. However, long context poses significant challenges for inference efficiency, including high memory costs of key-value (KV) cache and increased latency due to extensive memory accesses. Recent works have proposed compressing KV cache to approximate computation, but these methods either evict tokens permanently, never recalling them for later inference, or recall previous tokens at the granularity"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.03213","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.03213/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.03213","created_at":"2026-07-05T11:21:38.825915+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.03213v2","created_at":"2026-07-05T11:21:38.825915+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.03213","created_at":"2026-07-05T11:21:38.825915+00:00"},{"alias_kind":"pith_short_12","alias_value":"OSQTHM6WV4MT","created_at":"2026-07-05T11:21:38.825915+00:00"},{"alias_kind":"pith_short_16","alias_value":"OSQTHM6WV4MTWQFH","created_at":"2026-07-05T11:21:38.825915+00:00"},{"alias_kind":"pith_short_8","alias_value":"OSQTHM6W","created_at":"2026-07-05T11:21:38.825915+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19746","citing_title":"SAC: Disaggregated KV Cache System for Sparse Attention LLMs with CXL","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08446","citing_title":"Sparrow: Sparse Rollout for Stable and Efficient Long-context RL of Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23200","citing_title":"Adaptive Mass-Segmented KV Compression for Long-Context Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05772","citing_title":"Sparse Attention Remapping with Clustering for Efficient LLM Decoding on PIM","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17396","citing_title":"EpiCache: Episodic KV Cache Management for Long-Term Conversation on Resource-Constrained Environments","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2502.11089","citing_title":"Native Sparse Attention: Hardware-Aligned and Natively Trainable Sparse Attention","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12110","citing_title":"AB-Sparse: Sparse Attention with Adaptive Block Size for Accurate and Efficient Long-Context Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05365","citing_title":"ZAYA1-8B Technical Report","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18137","citing_title":"AQPIM: Breaking the PIM Capacity Wall for LLMs with In-Memory Activation Quantization","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3","json":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3.json","graph_json":"https://pith.science/api/pith-number/OSQTHM6WV4MTWQFHDA6FMK76Q3/graph.json","events_json":"https://pith.science/api/pith-number/OSQTHM6WV4MTWQFHDA6FMK76Q3/events.json","paper":"https://pith.science/paper/OSQTHM6W"},"agent_actions":{"view_html":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3","download_json":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3.json","view_paper":"https://pith.science/paper/OSQTHM6W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.03213&json=true","fetch_graph":"https://pith.science/api/pith-number/OSQTHM6WV4MTWQFHDA6FMK76Q3/graph.json","fetch_events":"https://pith.science/api/pith-number/OSQTHM6WV4MTWQFHDA6FMK76Q3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3/action/storage_attestation","attest_author":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3/action/author_attestation","sign_citation":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3/action/citation_signature","submit_replication":"https://pith.science/pith/OSQTHM6WV4MTWQFHDA6FMK76Q3/action/replication_record"}},"created_at":"2026-07-05T11:21:38.825915+00:00","updated_at":"2026-07-05T11:21:38.825915+00:00"}