{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KVV2YESPQFKR7L7HTPDG3JNAO2","short_pith_number":"pith:KVV2YESP","schema_version":"1.0","canonical_sha256":"556bac124f81551fafe79bc66da5a0768ca4a2bdcdeea939a8b905b3006a2968","source":{"kind":"arxiv","id":"2406.17565","version":3},"attestation_state":"computed","paper":{"title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chenxi Wang, Cunchen Hu, Heyang Huang, Jiang Xu, Junhao Hu, Ninghui Sun, Sa Wang, Tao Xie, Xusheng Chen, Yizhou Shan, Yungang Bao","submitted_at":"2024-06-25T14:02:08Z","abstract_excerpt":"Large language model (LLM) serving has transformed from stateless to stateful systems, utilizing techniques like context caching and disaggregated inference. These optimizations extend the lifespan and domain of the KV cache, necessitating a new architectural approach. We present MemServe, a unified system that integrates both inter-request and intra-request optimizations. MemServe introduces MemPool, an elastic memory pool managing distributed memory and KV caches across serving instances. Using MemPool APIs, MemServe combines context caching with disaggregated inference for the first time, s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17565","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-06-25T14:02:08Z","cross_cats_sorted":[],"title_canon_sha256":"2dfcf712d56fd4ebebc874d18091c47d8128636e64fb6afffecf7b20609e5d15","abstract_canon_sha256":"0cb0f202714a60936a0e00ee2a1a9d58fa0cc123ded1a3fc210bf58d591bc624"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:02.576652Z","signature_b64":"hbi/7BExUZ9sQA1KrVSb6WsPyhGJEpy+r77xQIIND+MncZLehW6fj2vaLLQS9nD2B8iVZHbdV8XXwnpQ+V+/Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"556bac124f81551fafe79bc66da5a0768ca4a2bdcdeea939a8b905b3006a2968","last_reissued_at":"2026-07-05T09:53:02.576182Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:02.576182Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Chenxi Wang, Cunchen Hu, Heyang Huang, Jiang Xu, Junhao Hu, Ninghui Sun, Sa Wang, Tao Xie, Xusheng Chen, Yizhou Shan, Yungang Bao","submitted_at":"2024-06-25T14:02:08Z","abstract_excerpt":"Large language model (LLM) serving has transformed from stateless to stateful systems, utilizing techniques like context caching and disaggregated inference. These optimizations extend the lifespan and domain of the KV cache, necessitating a new architectural approach. We present MemServe, a unified system that integrates both inter-request and intra-request optimizations. MemServe introduces MemPool, an elastic memory pool managing distributed memory and KV caches across serving instances. Using MemPool APIs, MemServe combines context caching with disaggregated inference for the first time, s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17565","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17565/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17565","created_at":"2026-07-05T09:53:02.576235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17565v3","created_at":"2026-07-05T09:53:02.576235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17565","created_at":"2026-07-05T09:53:02.576235+00:00"},{"alias_kind":"pith_short_12","alias_value":"KVV2YESPQFKR","created_at":"2026-07-05T09:53:02.576235+00:00"},{"alias_kind":"pith_short_16","alias_value":"KVV2YESPQFKR7L7H","created_at":"2026-07-05T09:53:02.576235+00:00"},{"alias_kind":"pith_short_8","alias_value":"KVV2YESP","created_at":"2026-07-05T09:53:02.576235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22541","citing_title":"ASAP: A Disaggregated and Asynchronous Inference System for MoE Prefill","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01065","citing_title":"Leyline: KV Cache Directives for Agentic Inference","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00866","citing_title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29207","citing_title":"KernelFlume: Elastic Core-Attention Scaling for Agentic Long-Context Decoding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23389","citing_title":"AlignedServe: Orchestrating Prefix-aware Batching to Build a High-throughput and Computing-efficient LLM Serving System","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22850","citing_title":"ObjectCache: Layerwise Object-Storage Retrieval for KV Cache Reuse","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2412.03594","citing_title":"BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20309","citing_title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09725","citing_title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27960","citing_title":"Towards Efficient Large Vision-Language Models: A Comprehensive Survey on Inference Strategies","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07173","citing_title":"InfiniLoRA: Disaggregated Multi-LoRA Serving for Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2","json":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2.json","graph_json":"https://pith.science/api/pith-number/KVV2YESPQFKR7L7HTPDG3JNAO2/graph.json","events_json":"https://pith.science/api/pith-number/KVV2YESPQFKR7L7HTPDG3JNAO2/events.json","paper":"https://pith.science/paper/KVV2YESP"},"agent_actions":{"view_html":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2","download_json":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2.json","view_paper":"https://pith.science/paper/KVV2YESP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17565&json=true","fetch_graph":"https://pith.science/api/pith-number/KVV2YESPQFKR7L7HTPDG3JNAO2/graph.json","fetch_events":"https://pith.science/api/pith-number/KVV2YESPQFKR7L7HTPDG3JNAO2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2/action/storage_attestation","attest_author":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2/action/author_attestation","sign_citation":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2/action/citation_signature","submit_replication":"https://pith.science/pith/KVV2YESPQFKR7L7HTPDG3JNAO2/action/replication_record"}},"created_at":"2026-07-05T09:53:02.576235+00:00","updated_at":"2026-07-05T09:53:02.576235+00:00"}