{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IOCPZKSVEYQEQEJ2HCRBYNEM4K","short_pith_number":"pith:IOCPZKSV","schema_version":"1.0","canonical_sha256":"4384fcaa55262048113a38a21c348ce29d855de7569b1534af87b32f49959574","source":{"kind":"arxiv","id":"2407.00023","version":2},"attestation_state":"computed","paper":{"title":"Preble: Efficient Distributed Prompt Scheduling for LLM Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Dongming Li, Reyna Abhyankar, Vikranth Srivatsa, Yiying Zhang, Zijian He","submitted_at":"2024-05-08T06:30:58Z","abstract_excerpt":"Prompts to large language models (LLMs) have evolved beyond simple user questions. For LLMs to solve complex problems, today's practices are to include domain-specific instructions, illustration of tool usages, and/or long context such as textbook chapters in prompts. As such, many parts of prompts are repetitive across requests. Recent works propose to cache and reuse KV state of prompts. However, they are all confined to a single-GPU optimization, while production LLM serving systems are distributed by nature.\n  This paper proposes Preble, the first distributed LLM serving platform that targ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00023","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-05-08T06:30:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"668e63cf501586ac15fc6579a8e139d4b04c6d4ae3fc2f12424ed84b7eb833de","abstract_canon_sha256":"38f0c0dd4755b86fe46aa00b3f1ed36ef5a5df12c841bb95517978e595913195"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:12.598401Z","signature_b64":"uQMcrNn4JYzbKEf2mpOQiYGIMOUfS8YbLAHK1qljvQOupOaIxBV0iqJTOBMiIl/E3z7MUn1tx8SC7piYVqVtDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4384fcaa55262048113a38a21c348ce29d855de7569b1534af87b32f49959574","last_reissued_at":"2026-07-05T09:15:12.597985Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:12.597985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Preble: Efficient Distributed Prompt Scheduling for LLM Serving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Dongming Li, Reyna Abhyankar, Vikranth Srivatsa, Yiying Zhang, Zijian He","submitted_at":"2024-05-08T06:30:58Z","abstract_excerpt":"Prompts to large language models (LLMs) have evolved beyond simple user questions. For LLMs to solve complex problems, today's practices are to include domain-specific instructions, illustration of tool usages, and/or long context such as textbook chapters in prompts. As such, many parts of prompts are repetitive across requests. Recent works propose to cache and reuse KV state of prompts. However, they are all confined to a single-GPU optimization, while production LLM serving systems are distributed by nature.\n  This paper proposes Preble, the first distributed LLM serving platform that targ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00023","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00023/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00023","created_at":"2026-07-05T09:15:12.598042+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00023v2","created_at":"2026-07-05T09:15:12.598042+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00023","created_at":"2026-07-05T09:15:12.598042+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOCPZKSVEYQE","created_at":"2026-07-05T09:15:12.598042+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOCPZKSVEYQEQEJ2","created_at":"2026-07-05T09:15:12.598042+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOCPZKSV","created_at":"2026-07-05T09:15:12.598042+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21238","citing_title":"Recency/Frequency Adaptive KV Caching for Large Language Model Serving","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18431","citing_title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00466","citing_title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00466","citing_title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31093","citing_title":"Omni-Flow: A Unified Workflow Orchestration and Distributed KV Cache Sharing Framework for Multimodal Inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16867","citing_title":"GoodServe: Towards High-Goodput Serving of Agentic LLM Inferences over Heterogeneous Resources","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09725","citing_title":"Efficient Remote KV Cache Reuse with GPU-native Video Codec","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24351","citing_title":"Diffusion Templates: A Unified Plugin Framework for Controllable Diffusion","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05219","citing_title":"Sparse Prefix Caching for Hybrid and Recurrent LLM Serving","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K","json":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K.json","graph_json":"https://pith.science/api/pith-number/IOCPZKSVEYQEQEJ2HCRBYNEM4K/graph.json","events_json":"https://pith.science/api/pith-number/IOCPZKSVEYQEQEJ2HCRBYNEM4K/events.json","paper":"https://pith.science/paper/IOCPZKSV"},"agent_actions":{"view_html":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K","download_json":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K.json","view_paper":"https://pith.science/paper/IOCPZKSV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00023&json=true","fetch_graph":"https://pith.science/api/pith-number/IOCPZKSVEYQEQEJ2HCRBYNEM4K/graph.json","fetch_events":"https://pith.science/api/pith-number/IOCPZKSVEYQEQEJ2HCRBYNEM4K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K/action/storage_attestation","attest_author":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K/action/author_attestation","sign_citation":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K/action/citation_signature","submit_replication":"https://pith.science/pith/IOCPZKSVEYQEQEJ2HCRBYNEM4K/action/replication_record"}},"created_at":"2026-07-05T09:15:12.598042+00:00","updated_at":"2026-07-05T09:15:12.598042+00:00"}