{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T6BIF3DLRCOQPH372ATXVAZKBV","short_pith_number":"pith:T6BIF3DL","schema_version":"1.0","canonical_sha256":"9f8282ec6b889d079f7fd0277a832a0d625db0dae32a327b1409e83f2bd04cc5","source":{"kind":"arxiv","id":"2410.21465","version":3},"attestation_state":"computed","paper":{"title":"ShadowKV: KV Cache in Shadows for High-Throughput Long-Context LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Beidi Chen, Hanshi Sun, Harry Dong, Li-wen Chang, Ningxin Zheng, Size Zheng, Wenlei Bao, Xin Liu, Yuejie Chi","submitted_at":"2024-10-28T19:08:12Z","abstract_excerpt":"With the widespread deployment of long-context large language models (LLMs), there has been a growing demand for efficient support of high-throughput inference. However, as the key-value (KV) cache expands with the sequence length, the increasing memory footprint and the need to access it for each token generation both result in low throughput when serving long-context LLMs. While various dynamic sparse attention methods have been proposed to speed up inference while maintaining generation quality, they either fail to sufficiently reduce GPU memory consumption or introduce significant decoding"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21465","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-28T19:08:12Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d2ca96f1dcddb4eaead33c78cc6f0cf313d02eecaa5f6024ab639fa66eb46df9","abstract_canon_sha256":"2799d9364e51ae5fcd13d3aa84dc34487f536b54ca8d83888c0f6cd3c4a23866"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:13.663010Z","signature_b64":"hRruuOBepDg3JAW7/+y2/yJ9Pa/pxaOKNOUfavmv5W6usZi92vWsaeNKxeAOTgF/xisdimIfiOX9UEzuI3+aAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f8282ec6b889d079f7fd0277a832a0d625db0dae32a327b1409e83f2bd04cc5","last_reissued_at":"2026-07-05T10:54:13.662504Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:13.662504Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ShadowKV: KV Cache in Shadows for High-Throughput Long-Context LLM Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Beidi Chen, Hanshi Sun, Harry Dong, Li-wen Chang, Ningxin Zheng, Size Zheng, Wenlei Bao, Xin Liu, Yuejie Chi","submitted_at":"2024-10-28T19:08:12Z","abstract_excerpt":"With the widespread deployment of long-context large language models (LLMs), there has been a growing demand for efficient support of high-throughput inference. However, as the key-value (KV) cache expands with the sequence length, the increasing memory footprint and the need to access it for each token generation both result in low throughput when serving long-context LLMs. While various dynamic sparse attention methods have been proposed to speed up inference while maintaining generation quality, they either fail to sufficiently reduce GPU memory consumption or introduce significant decoding"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21465","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21465/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21465","created_at":"2026-07-05T10:54:13.662568+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21465v3","created_at":"2026-07-05T10:54:13.662568+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21465","created_at":"2026-07-05T10:54:13.662568+00:00"},{"alias_kind":"pith_short_12","alias_value":"T6BIF3DLRCOQ","created_at":"2026-07-05T10:54:13.662568+00:00"},{"alias_kind":"pith_short_16","alias_value":"T6BIF3DLRCOQPH37","created_at":"2026-07-05T10:54:13.662568+00:00"},{"alias_kind":"pith_short_8","alias_value":"T6BIF3DL","created_at":"2026-07-05T10:54:13.662568+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":109,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17708","citing_title":"Co-evolving Agent Architectures and Interpretable Reasoning for Automated Optimization","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06453","citing_title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03825","citing_title":"Dynamic Short Convolutions Improve Transformers","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00866","citing_title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30389","citing_title":"Predict, Reuse, and Repair: Accelerating Dynamic Sparse Attention for Long-Context LLM Decoding","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10726","citing_title":"PrefixWall: Mitigating Prefix Caching Side Channels in Shared LLM Systems","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18753","citing_title":"DashAttention: Differentiable and Adaptive Sparse Hierarchical Attention","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21623","citing_title":"OjaKV: Context-Aware Online Low-Rank KV Cache Compression","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13684","citing_title":"HeteroCache: A Dynamic Retrieval Approach to Heterogeneous KV Cache Compression for Long-Context LLM Inference","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09649","citing_title":"Make Each Token Count: Towards Improving Long-Context Performance with KV Cache Eviction","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07719","citing_title":"An Efficient Hybrid Sparse Attention with CPU-GPU Parallelism for Long-Context Inference","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17709","citing_title":"DeInfer: Efficient Parallel Inferencing for Decomposed Large Language Models","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV","json":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV.json","graph_json":"https://pith.science/api/pith-number/T6BIF3DLRCOQPH372ATXVAZKBV/graph.json","events_json":"https://pith.science/api/pith-number/T6BIF3DLRCOQPH372ATXVAZKBV/events.json","paper":"https://pith.science/paper/T6BIF3DL"},"agent_actions":{"view_html":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV","download_json":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV.json","view_paper":"https://pith.science/paper/T6BIF3DL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21465&json=true","fetch_graph":"https://pith.science/api/pith-number/T6BIF3DLRCOQPH372ATXVAZKBV/graph.json","fetch_events":"https://pith.science/api/pith-number/T6BIF3DLRCOQPH372ATXVAZKBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV/action/storage_attestation","attest_author":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV/action/author_attestation","sign_citation":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV/action/citation_signature","submit_replication":"https://pith.science/pith/T6BIF3DLRCOQPH372ATXVAZKBV/action/replication_record"}},"created_at":"2026-07-05T10:54:13.662568+00:00","updated_at":"2026-07-05T10:54:13.662568+00:00"}