{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LBFAFEU2P3R6NTENPARM4DZYLM","short_pith_number":"pith:LBFAFEU2","schema_version":"1.0","canonical_sha256":"584a02929a7ee3e6cc8d7822ce0f385b259a9065870ad9afe4b471207b91b522","source":{"kind":"arxiv","id":"2406.03243","version":1},"attestation_state":"computed","paper":{"title":"Llumnix: Dynamic Scheduling for Large Language Model Serving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.LG"],"primary_cat":"cs.AR","authors_text":"Biao Sun, Hanyu Zhao, Wei Lin, Wencong Xiao, Xinyi Zhang, Yong Li, Ziming Huang","submitted_at":"2024-06-05T13:20:18Z","abstract_excerpt":"Inference serving for large language models (LLMs) is the key to unleashing their potential in people's daily lives. However, efficient LLM serving remains challenging today because the requests are inherently heterogeneous and unpredictable in terms of resource and latency requirements, as a result of the diverse applications and the dynamic execution nature of LLMs. Existing systems are fundamentally limited in handling these characteristics and cause problems such as severe queuing delays, poor tail latencies, and SLO violations.\n  We introduce Llumnix, an LLM serving system that reacts to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03243","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AR","submitted_at":"2024-06-05T13:20:18Z","cross_cats_sorted":["cs.DC","cs.LG"],"title_canon_sha256":"cb678e8f863a029cb9306ed7d729040202aaf3fd5b82618592b4df14543beb5a","abstract_canon_sha256":"546d4ada129c962b01315a76fabd7d07842f21db3216125926a4fe99e7f10c2a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:12.027639Z","signature_b64":"nEHCjHkCinT0jF9NYN1YNoeyoGeNzWTWirN9xJXQ1N5F9uf2OLtmj0+huY7gJ6Hzsl5cDafWMzsAaBYGbmInCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"584a02929a7ee3e6cc8d7822ce0f385b259a9065870ad9afe4b471207b91b522","last_reissued_at":"2026-07-05T08:28:12.027211Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:12.027211Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Llumnix: Dynamic Scheduling for Large Language Model Serving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","cs.LG"],"primary_cat":"cs.AR","authors_text":"Biao Sun, Hanyu Zhao, Wei Lin, Wencong Xiao, Xinyi Zhang, Yong Li, Ziming Huang","submitted_at":"2024-06-05T13:20:18Z","abstract_excerpt":"Inference serving for large language models (LLMs) is the key to unleashing their potential in people's daily lives. However, efficient LLM serving remains challenging today because the requests are inherently heterogeneous and unpredictable in terms of resource and latency requirements, as a result of the diverse applications and the dynamic execution nature of LLMs. Existing systems are fundamentally limited in handling these characteristics and cause problems such as severe queuing delays, poor tail latencies, and SLO violations.\n  We introduce Llumnix, an LLM serving system that reacts to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03243","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03243/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03243","created_at":"2026-07-05T08:28:12.027267+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03243v1","created_at":"2026-07-05T08:28:12.027267+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03243","created_at":"2026-07-05T08:28:12.027267+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBFAFEU2P3R6","created_at":"2026-07-05T08:28:12.027267+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBFAFEU2P3R6NTEN","created_at":"2026-07-05T08:28:12.027267+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBFAFEU2","created_at":"2026-07-05T08:28:12.027267+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":209,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01579","citing_title":"OmniPilot: An Uncertainty-Aware LLM Inference Advisor for Heterogeneous GPU Clusters","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20863","citing_title":"PlexRL: Cluster-Level Orchestration of Serviceized LLM Execution for RLVR","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":288,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04750","citing_title":"DeepStack: Scalable and Accurate Design Space Exploration for Distributed 3D-Stacked AI Accelerators","ref_index":100,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM","json":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM.json","graph_json":"https://pith.science/api/pith-number/LBFAFEU2P3R6NTENPARM4DZYLM/graph.json","events_json":"https://pith.science/api/pith-number/LBFAFEU2P3R6NTENPARM4DZYLM/events.json","paper":"https://pith.science/paper/LBFAFEU2"},"agent_actions":{"view_html":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM","download_json":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM.json","view_paper":"https://pith.science/paper/LBFAFEU2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03243&json=true","fetch_graph":"https://pith.science/api/pith-number/LBFAFEU2P3R6NTENPARM4DZYLM/graph.json","fetch_events":"https://pith.science/api/pith-number/LBFAFEU2P3R6NTENPARM4DZYLM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM/action/storage_attestation","attest_author":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM/action/author_attestation","sign_citation":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM/action/citation_signature","submit_replication":"https://pith.science/pith/LBFAFEU2P3R6NTENPARM4DZYLM/action/replication_record"}},"created_at":"2026-07-05T08:28:12.027267+00:00","updated_at":"2026-07-05T08:28:12.027267+00:00"}