{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VSQFCPLSRBPFGKHE6TT2U6CTJE","short_pith_number":"pith:VSQFCPLS","schema_version":"1.0","canonical_sha256":"aca0513d72885e5328e4f4e7aa7853491d6f01cea8d2491d361b3805022c49d5","source":{"kind":"arxiv","id":"2404.08509","version":2},"attestation_state":"computed","paper":{"title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.DC","authors_text":"Archit Patke, Chen Wang, Haoran Qiu, Hubertus Franke, Ravishankar K. Iyer, Saurabh Jha, Shengkun Cui, Tamer Ba\\c{s}ar, Weichao Mao, Zbigniew T. Kalbarczyk","submitted_at":"2024-04-12T14:46:15Z","abstract_excerpt":"Large language models (LLMs) have been driving a new wave of interactive AI applications across numerous domains. However, efficiently serving LLM inference requests is challenging due to their unpredictable execution times originating from the autoregressive nature of generative models. Existing LLM serving systems exploit first-come-first-serve (FCFS) scheduling, suffering from head-of-line blocking issues. To address the non-deterministic nature of LLMs and enable efficient interactive LLM serving, we present a speculative shortest-job-first (SSJF) scheduler that uses a light proxy model to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.08509","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-04-12T14:46:15Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"573bcc6daf20d4f70877feb4cc29dc80846553839de2d0601ff16567b77dae53","abstract_canon_sha256":"1c304349f937499bad435658099bb14c8bb6c96a5f22a0bc79f829e342e76980"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:54.180174Z","signature_b64":"dxYu3gn1D4k5qdrSrYxYknNvdDxiQHmv2mE8ijPPtU2/0SBH7xAhBHmgnicaI8M+rPxZegNf06miMyxO4aXPBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aca0513d72885e5328e4f4e7aa7853491d6f01cea8d2491d361b3805022c49d5","last_reissued_at":"2026-07-05T09:39:54.179561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:54.179561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.DC","authors_text":"Archit Patke, Chen Wang, Haoran Qiu, Hubertus Franke, Ravishankar K. Iyer, Saurabh Jha, Shengkun Cui, Tamer Ba\\c{s}ar, Weichao Mao, Zbigniew T. Kalbarczyk","submitted_at":"2024-04-12T14:46:15Z","abstract_excerpt":"Large language models (LLMs) have been driving a new wave of interactive AI applications across numerous domains. However, efficiently serving LLM inference requests is challenging due to their unpredictable execution times originating from the autoregressive nature of generative models. Existing LLM serving systems exploit first-come-first-serve (FCFS) scheduling, suffering from head-of-line blocking issues. To address the non-deterministic nature of LLMs and enable efficient interactive LLM serving, we present a speculative shortest-job-first (SSJF) scheduler that uses a light proxy model to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08509","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.08509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.08509","created_at":"2026-07-05T09:39:54.179629+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.08509v2","created_at":"2026-07-05T09:39:54.179629+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08509","created_at":"2026-07-05T09:39:54.179629+00:00"},{"alias_kind":"pith_short_12","alias_value":"VSQFCPLSRBPF","created_at":"2026-07-05T09:39:54.179629+00:00"},{"alias_kind":"pith_short_16","alias_value":"VSQFCPLSRBPFGKHE","created_at":"2026-07-05T09:39:54.179629+00:00"},{"alias_kind":"pith_short_8","alias_value":"VSQFCPLS","created_at":"2026-07-05T09:39:54.179629+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22327","citing_title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18431","citing_title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07248","citing_title":"Clairvoyant: Predictive Shortest-Job-First Admission for Serial LLM Inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30391","citing_title":"Energy-Aware Scheduling for Serverless LLM Serving on Shared GPUs","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19179","citing_title":"CascadeInfer: Length-Aware Scheduling of LLM Serving with Low Latency and Load Balancing","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20309","citing_title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13668","citing_title":"STAR: Decode-Phase Rescheduling for LLM Inference","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06113","citing_title":"Tackling the Data-Parallel Load Balancing Bottleneck in LLM Serving: Practical Online Routing at Scale","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04595","citing_title":"A Queueing-Theoretic Framework for Stability Analysis of LLM Inference with KV Cache Memory Constraints","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07144","citing_title":"Autopoiesis: A Self-Evolving System Paradigm for LLM Serving Under Runtime Dynamics","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06113","citing_title":"Tackling the Data-Parallel Load Balancing Bottleneck in LLM Serving: Practical Online Routing at Scale","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07931","citing_title":"Robust Length Prediction: A Perspective from Heavy-Tailed Prompt-Conditioned Distributions","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE","json":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE.json","graph_json":"https://pith.science/api/pith-number/VSQFCPLSRBPFGKHE6TT2U6CTJE/graph.json","events_json":"https://pith.science/api/pith-number/VSQFCPLSRBPFGKHE6TT2U6CTJE/events.json","paper":"https://pith.science/paper/VSQFCPLS"},"agent_actions":{"view_html":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE","download_json":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE.json","view_paper":"https://pith.science/paper/VSQFCPLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.08509&json=true","fetch_graph":"https://pith.science/api/pith-number/VSQFCPLSRBPFGKHE6TT2U6CTJE/graph.json","fetch_events":"https://pith.science/api/pith-number/VSQFCPLSRBPFGKHE6TT2U6CTJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE/action/storage_attestation","attest_author":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE/action/author_attestation","sign_citation":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE/action/citation_signature","submit_replication":"https://pith.science/pith/VSQFCPLSRBPFGKHE6TT2U6CTJE/action/replication_record"}},"created_at":"2026-07-05T09:39:54.179629+00:00","updated_at":"2026-07-05T09:39:54.179629+00:00"}