{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IGCC3R36GVF6FXS27MCXDS5IEV","short_pith_number":"pith:IGCC3R36","schema_version":"1.0","canonical_sha256":"41842dc77e354be2de5afb0571cba8254706d63012fa1b3534ed67091abc2e6f","source":{"kind":"arxiv","id":"2410.00428","version":3},"attestation_state":"computed","paper":{"title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Changxu Shao, Hao Wu, Junping Zhao, Ke Zhang, Rui Zhang, Yi Xiong, Yuhong Guo, Zhenxuan Pan, Ziqing Wang","submitted_at":"2024-10-01T06:23:17Z","abstract_excerpt":"The expanding context windows in large language models (LLMs) have greatly enhanced their capabilities in various applications, but they also introduce significant challenges in maintaining low latency, particularly in Time to First Token (TTFT). This paper identifies that the sharp rise in TTFT as context length increases is predominantly driven by queuing delays, which are caused by the growing demands for GPU Key-Value (KV) cache allocation clashing with the limited availability of KV cache blocks. To address this issue, we propose LayerKV, a simple yet effective plug-in method that effecti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.00428","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.DC","submitted_at":"2024-10-01T06:23:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2f8e23be42f7832896eca08281a466bd5abea6d3da6732a8f3d54fbe9d37836c","abstract_canon_sha256":"face6046d5c0feb0ef0180bd7a953beae046722214352c57fb76ec0b177ccf25"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:49.749779Z","signature_b64":"4HUydmEl6JTXnKA88pey321c/QYsCACgF1JnE83s6xICjPgFMDu5uiZ6ywhF5lTiMXimNiA/9Ph0mNsEsmYRDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41842dc77e354be2de5afb0571cba8254706d63012fa1b3534ed67091abc2e6f","last_reissued_at":"2026-07-05T09:17:49.749256Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:49.749256Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LayerKV: Optimizing Large Language Model Serving with Layer-wise KV Cache Management","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Changxu Shao, Hao Wu, Junping Zhao, Ke Zhang, Rui Zhang, Yi Xiong, Yuhong Guo, Zhenxuan Pan, Ziqing Wang","submitted_at":"2024-10-01T06:23:17Z","abstract_excerpt":"The expanding context windows in large language models (LLMs) have greatly enhanced their capabilities in various applications, but they also introduce significant challenges in maintaining low latency, particularly in Time to First Token (TTFT). This paper identifies that the sharp rise in TTFT as context length increases is predominantly driven by queuing delays, which are caused by the growing demands for GPU Key-Value (KV) cache allocation clashing with the limited availability of KV cache blocks. To address this issue, we propose LayerKV, a simple yet effective plug-in method that effecti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.00428","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.00428/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.00428","created_at":"2026-07-05T09:17:49.749323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.00428v3","created_at":"2026-07-05T09:17:49.749323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.00428","created_at":"2026-07-05T09:17:49.749323+00:00"},{"alias_kind":"pith_short_12","alias_value":"IGCC3R36GVF6","created_at":"2026-07-05T09:17:49.749323+00:00"},{"alias_kind":"pith_short_16","alias_value":"IGCC3R36GVF6FXS2","created_at":"2026-07-05T09:17:49.749323+00:00"},{"alias_kind":"pith_short_8","alias_value":"IGCC3R36","created_at":"2026-07-05T09:17:49.749323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24022","citing_title":"Adaptive KV Cache Reuse for Fast Long-Context LLM Serving","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09725","citing_title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06370","citing_title":"ForkKV: Scaling Multi-LoRA Agent Serving via Copy-on-Write Disaggregated KV Cache","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV","json":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV.json","graph_json":"https://pith.science/api/pith-number/IGCC3R36GVF6FXS27MCXDS5IEV/graph.json","events_json":"https://pith.science/api/pith-number/IGCC3R36GVF6FXS27MCXDS5IEV/events.json","paper":"https://pith.science/paper/IGCC3R36"},"agent_actions":{"view_html":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV","download_json":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV.json","view_paper":"https://pith.science/paper/IGCC3R36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.00428&json=true","fetch_graph":"https://pith.science/api/pith-number/IGCC3R36GVF6FXS27MCXDS5IEV/graph.json","fetch_events":"https://pith.science/api/pith-number/IGCC3R36GVF6FXS27MCXDS5IEV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV/action/storage_attestation","attest_author":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV/action/author_attestation","sign_citation":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV/action/citation_signature","submit_replication":"https://pith.science/pith/IGCC3R36GVF6FXS27MCXDS5IEV/action/replication_record"}},"created_at":"2026-07-05T09:17:49.749323+00:00","updated_at":"2026-07-05T09:17:49.749323+00:00"}