{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MFS6ILLFYLHVSGJW3PZYIOVBHP","short_pith_number":"pith:MFS6ILLF","schema_version":"1.0","canonical_sha256":"6165e42d65c2cf591936dbf3843aa13bee740e9f2c3198c7962868e1827a32e1","source":{"kind":"arxiv","id":"2501.12162","version":2},"attestation_state":"computed","paper":{"title":"AdaServe: Accelerating Multi-SLO LLM Serving with SLO-Customized Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.LG"],"primary_cat":"cs.CL","authors_text":"April Yang, Gabriele Oliaro, Qinghan Chen, Remi Delacourt, Sean Lai, Shuhuai Lin, Xinhao Cheng, Xupeng Miao, Zeyu Wang, Zhihao Jia, Zhihao Zhang, Zhuofu Chen, Zhuoming Chen, Zikun Li","submitted_at":"2025-01-21T14:15:01Z","abstract_excerpt":"Modern large language model (LLM) applications exhibit diverse service-level objectives (SLOs), from low-latency requirements in interactive coding assistants to more relaxed constraints in data wrangling tasks. Existing LLM serving systems, which rely on uniform batching and scheduling strategies, often fail to meet these heterogeneous SLOs concurrently. We present AdaServe, the first LLM serving system designed to support efficient multi-SLO serving through SLO-customized speculative decoding. AdaServe formulates multi-SLO serving as a constrained optimization problem and introduces a hardwa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.12162","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-21T14:15:01Z","cross_cats_sorted":["cs.AI","cs.DC","cs.LG"],"title_canon_sha256":"b06c96127cc9c4ee6ab88398a346a145e29917cdb2ff67fbd6e65d921f50505b","abstract_canon_sha256":"63cf6b5746996c92899466160db5a05b9e396207e84088349652761b42b5d9a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:44.262669Z","signature_b64":"h2acz6W8QLys1I4Ldas8UE46bL2BE0a4lTeWxvyOa0gbAC55djXM6eS4zK7U6hsBNeutWRFVVR02buFnnco9Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6165e42d65c2cf591936dbf3843aa13bee740e9f2c3198c7962868e1827a32e1","last_reissued_at":"2026-07-05T11:04:44.262168Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:44.262168Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaServe: Accelerating Multi-SLO LLM Serving with SLO-Customized Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC","cs.LG"],"primary_cat":"cs.CL","authors_text":"April Yang, Gabriele Oliaro, Qinghan Chen, Remi Delacourt, Sean Lai, Shuhuai Lin, Xinhao Cheng, Xupeng Miao, Zeyu Wang, Zhihao Jia, Zhihao Zhang, Zhuofu Chen, Zhuoming Chen, Zikun Li","submitted_at":"2025-01-21T14:15:01Z","abstract_excerpt":"Modern large language model (LLM) applications exhibit diverse service-level objectives (SLOs), from low-latency requirements in interactive coding assistants to more relaxed constraints in data wrangling tasks. Existing LLM serving systems, which rely on uniform batching and scheduling strategies, often fail to meet these heterogeneous SLOs concurrently. We present AdaServe, the first LLM serving system designed to support efficient multi-SLO serving through SLO-customized speculative decoding. AdaServe formulates multi-SLO serving as a constrained optimization problem and introduces a hardwa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.12162","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.12162/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.12162","created_at":"2026-07-05T11:04:44.262232+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.12162v2","created_at":"2026-07-05T11:04:44.262232+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.12162","created_at":"2026-07-05T11:04:44.262232+00:00"},{"alias_kind":"pith_short_12","alias_value":"MFS6ILLFYLHV","created_at":"2026-07-05T11:04:44.262232+00:00"},{"alias_kind":"pith_short_16","alias_value":"MFS6ILLFYLHVSGJW","created_at":"2026-07-05T11:04:44.262232+00:00"},{"alias_kind":"pith_short_8","alias_value":"MFS6ILLF","created_at":"2026-07-05T11:04:44.262232+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24832","citing_title":"Optimus: Elastic Decoding for Efficient Diffusion LLM Serving","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20309","citing_title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04357","citing_title":"Coral: Cost-Efficient Multi-LLM Serving over Heterogeneous Cloud GPUs","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06914","citing_title":"Regulating Branch Parallelism in LLM Serving","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17227","citing_title":"Cloud-native and Distributed Systems for Efficient and Scalable Large Language Models -- A Research Agenda","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20503","citing_title":"FASER: Fine-Grained Phase Management for Speculative Decoding in Dynamic LLM Serving","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP","json":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP.json","graph_json":"https://pith.science/api/pith-number/MFS6ILLFYLHVSGJW3PZYIOVBHP/graph.json","events_json":"https://pith.science/api/pith-number/MFS6ILLFYLHVSGJW3PZYIOVBHP/events.json","paper":"https://pith.science/paper/MFS6ILLF"},"agent_actions":{"view_html":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP","download_json":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP.json","view_paper":"https://pith.science/paper/MFS6ILLF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.12162&json=true","fetch_graph":"https://pith.science/api/pith-number/MFS6ILLFYLHVSGJW3PZYIOVBHP/graph.json","fetch_events":"https://pith.science/api/pith-number/MFS6ILLFYLHVSGJW3PZYIOVBHP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP/action/storage_attestation","attest_author":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP/action/author_attestation","sign_citation":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP/action/citation_signature","submit_replication":"https://pith.science/pith/MFS6ILLFYLHVSGJW3PZYIOVBHP/action/replication_record"}},"created_at":"2026-07-05T11:04:44.262232+00:00","updated_at":"2026-07-05T11:04:44.262232+00:00"}