{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:N64Y3K2SJGS6U4CKWM5ATTYHTN","short_pith_number":"pith:N64Y3K2S","schema_version":"1.0","canonical_sha256":"6fb98dab5249a5ea704ab33a09cf079b5127150c6f3d68cbdbbd7d2404369fc3","source":{"kind":"arxiv","id":"2410.01035","version":1},"attestation_state":"computed","paper":{"title":"Don't Stop Me Now: Embedding Based Scheduling for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chunwei Liu, Eran Malach, Michael Mitzenmacher, Minlan Yu, Rana Shahout, Weifan Jiang","submitted_at":"2024-10-01T19:51:07Z","abstract_excerpt":"Efficient scheduling is crucial for interactive Large Language Model (LLM) applications, where low request completion time directly impacts user engagement. Size-based scheduling algorithms like Shortest Remaining Process Time (SRPT) aim to reduce average request completion time by leveraging known or estimated request sizes and allowing preemption by incoming jobs with shorter service times. However, two main challenges arise when applying size-based scheduling to LLM systems. First, accurately predicting output lengths from prompts is challenging and often resource-intensive, making it impra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01035","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-01T19:51:07Z","cross_cats_sorted":[],"title_canon_sha256":"882c6d02c795ea8f76dad47be433bbef00cf6ac515019e5e32eb0c50bbb7c28d","abstract_canon_sha256":"be385a3e7a1f545cb0ee573b26ab08d11f8b77778c6425ac2500200d1dddbd38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:42.155092Z","signature_b64":"B9BxjJHvDInQiH9e+SKO51bMQ/lbD+xNzlY2K7Vlc1vh1jcqFi/XbLqj0Upf/G5QkE0GsfAsLe4GZjiqMHZLDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6fb98dab5249a5ea704ab33a09cf079b5127150c6f3d68cbdbbd7d2404369fc3","last_reissued_at":"2026-07-05T09:14:42.154658Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:42.154658Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't Stop Me Now: Embedding Based Scheduling for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chunwei Liu, Eran Malach, Michael Mitzenmacher, Minlan Yu, Rana Shahout, Weifan Jiang","submitted_at":"2024-10-01T19:51:07Z","abstract_excerpt":"Efficient scheduling is crucial for interactive Large Language Model (LLM) applications, where low request completion time directly impacts user engagement. Size-based scheduling algorithms like Shortest Remaining Process Time (SRPT) aim to reduce average request completion time by leveraging known or estimated request sizes and allowing preemption by incoming jobs with shorter service times. However, two main challenges arise when applying size-based scheduling to LLM systems. First, accurately predicting output lengths from prompts is challenging and often resource-intensive, making it impra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01035","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01035/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01035","created_at":"2026-07-05T09:14:42.154719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01035v1","created_at":"2026-07-05T09:14:42.154719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01035","created_at":"2026-07-05T09:14:42.154719+00:00"},{"alias_kind":"pith_short_12","alias_value":"N64Y3K2SJGS6","created_at":"2026-07-05T09:14:42.154719+00:00"},{"alias_kind":"pith_short_16","alias_value":"N64Y3K2SJGS6U4CK","created_at":"2026-07-05T09:14:42.154719+00:00"},{"alias_kind":"pith_short_8","alias_value":"N64Y3K2S","created_at":"2026-07-05T09:14:42.154719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22327","citing_title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18431","citing_title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21427","citing_title":"PALS: Power-Aware LLM Serving for Mixture-of-Experts Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06113","citing_title":"Tackling the Data-Parallel Load Balancing Bottleneck in LLM Serving: Practical Online Routing at Scale","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04595","citing_title":"A Queueing-Theoretic Framework for Stability Analysis of LLM Inference with KV Cache Memory Constraints","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01280","citing_title":"Position: LLM Serving Needs Mathematical Optimization and Algorithmic Foundations, Not Just Heuristics","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06113","citing_title":"Tackling the Data-Parallel Load Balancing Bottleneck in LLM Serving: Practical Online Routing at Scale","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN","json":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN.json","graph_json":"https://pith.science/api/pith-number/N64Y3K2SJGS6U4CKWM5ATTYHTN/graph.json","events_json":"https://pith.science/api/pith-number/N64Y3K2SJGS6U4CKWM5ATTYHTN/events.json","paper":"https://pith.science/paper/N64Y3K2S"},"agent_actions":{"view_html":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN","download_json":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN.json","view_paper":"https://pith.science/paper/N64Y3K2S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01035&json=true","fetch_graph":"https://pith.science/api/pith-number/N64Y3K2SJGS6U4CKWM5ATTYHTN/graph.json","fetch_events":"https://pith.science/api/pith-number/N64Y3K2SJGS6U4CKWM5ATTYHTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN/action/storage_attestation","attest_author":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN/action/author_attestation","sign_citation":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN/action/citation_signature","submit_replication":"https://pith.science/pith/N64Y3K2SJGS6U4CKWM5ATTYHTN/action/replication_record"}},"created_at":"2026-07-05T09:14:42.154719+00:00","updated_at":"2026-07-05T09:14:42.154719+00:00"}