{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RTWAC2I7B5AUKAO4ORCBZ5YYF4","short_pith_number":"pith:RTWAC2I7","schema_version":"1.0","canonical_sha256":"8cec01691f0f414501dc74441cf7182f0cabbbfd56978905afaa26431529e2f1","source":{"kind":"arxiv","id":"2505.09142","version":1},"attestation_state":"computed","paper":{"title":"ELIS: Efficient LLM Iterative Scheduling System with Response Length Predictor","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Eunjoo Jeon, Jeonghoe Goo, Mingyu Yang, Minsung Jang, Seungbeom Choi","submitted_at":"2025-05-14T04:50:00Z","abstract_excerpt":"We propose ELIS, a serving system for Large Language Models (LLMs) featuring an Iterative Shortest Remaining Time First (ISRTF) scheduler designed to efficiently manage inference tasks with the shortest remaining tokens. Current LLM serving systems often employ a first-come-first-served scheduling strategy, which can lead to the \"head-of-line blocking\" problem. To overcome this limitation, it is necessary to predict LLM inference times and apply a shortest job first scheduling strategy. However, due to the auto-regressive nature of LLMs, predicting the inference latency is challenging. ELIS ad"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.09142","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-05-14T04:50:00Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"fa9dc5ed25aea41c1690159f778a1d47af4fd2e9bfd1d1e91f09c34e88c5bf24","abstract_canon_sha256":"171e746baf79a6d7d210ba522cbf70cd813203779681666e079ec6c7a0c21df6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:50.934408Z","signature_b64":"kOmLMDSVOEUVuAoLdcIG/pRewF/tTV+97HC1NYN5Yukb14Cw/iWLawXzOsB1b5TYMxiVWJxAG1qRNip/zE5EAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8cec01691f0f414501dc74441cf7182f0cabbbfd56978905afaa26431529e2f1","last_reissued_at":"2026-07-05T11:02:50.933917Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:50.933917Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ELIS: Efficient LLM Iterative Scheduling System with Response Length Predictor","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Eunjoo Jeon, Jeonghoe Goo, Mingyu Yang, Minsung Jang, Seungbeom Choi","submitted_at":"2025-05-14T04:50:00Z","abstract_excerpt":"We propose ELIS, a serving system for Large Language Models (LLMs) featuring an Iterative Shortest Remaining Time First (ISRTF) scheduler designed to efficiently manage inference tasks with the shortest remaining tokens. Current LLM serving systems often employ a first-come-first-served scheduling strategy, which can lead to the \"head-of-line blocking\" problem. To overcome this limitation, it is necessary to predict LLM inference times and apply a shortest job first scheduling strategy. However, due to the auto-regressive nature of LLMs, predicting the inference latency is challenging. ELIS ad"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.09142","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.09142/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.09142","created_at":"2026-07-05T11:02:50.933974+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.09142v1","created_at":"2026-07-05T11:02:50.933974+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.09142","created_at":"2026-07-05T11:02:50.933974+00:00"},{"alias_kind":"pith_short_12","alias_value":"RTWAC2I7B5AU","created_at":"2026-07-05T11:02:50.933974+00:00"},{"alias_kind":"pith_short_16","alias_value":"RTWAC2I7B5AUKAO4","created_at":"2026-07-05T11:02:50.933974+00:00"},{"alias_kind":"pith_short_8","alias_value":"RTWAC2I7","created_at":"2026-07-05T11:02:50.933974+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30391","citing_title":"Energy-Aware Scheduling for Serverless LLM Serving on Shared GPUs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09002","citing_title":"Security Considerations for Multi-agent Systems","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07931","citing_title":"Robust Length Prediction: A Perspective from Heavy-Tailed Prompt-Conditioned Distributions","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4","json":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4.json","graph_json":"https://pith.science/api/pith-number/RTWAC2I7B5AUKAO4ORCBZ5YYF4/graph.json","events_json":"https://pith.science/api/pith-number/RTWAC2I7B5AUKAO4ORCBZ5YYF4/events.json","paper":"https://pith.science/paper/RTWAC2I7"},"agent_actions":{"view_html":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4","download_json":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4.json","view_paper":"https://pith.science/paper/RTWAC2I7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.09142&json=true","fetch_graph":"https://pith.science/api/pith-number/RTWAC2I7B5AUKAO4ORCBZ5YYF4/graph.json","fetch_events":"https://pith.science/api/pith-number/RTWAC2I7B5AUKAO4ORCBZ5YYF4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4/action/storage_attestation","attest_author":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4/action/author_attestation","sign_citation":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4/action/citation_signature","submit_replication":"https://pith.science/pith/RTWAC2I7B5AUKAO4ORCBZ5YYF4/action/replication_record"}},"created_at":"2026-07-05T11:02:50.933974+00:00","updated_at":"2026-07-05T11:02:50.933974+00:00"}