{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KWPBK6DA3LV5YOWT5VX5FNGMYQ","short_pith_number":"pith:KWPBK6DA","schema_version":"1.0","canonical_sha256":"559e157860daebdc3ad3ed6fd2b4ccc405c07399009606c8c8416fe92c4cd5b6","source":{"kind":"arxiv","id":"2502.13965","version":1},"attestation_state":"computed","paper":{"title":"Autellix: An Efficient Serving Engine for LLM Agents as General Programs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC"],"primary_cat":"cs.LG","authors_text":"Chi Wang, Colin Cai, Ion Stoica, Joseph E. Gonzalez, Justin Wong, Michael Luo, Tianjun Zhang, Xiaoxiang Shi, Yanping Huang, Yichuan Wang, Zhifeng Chen","submitted_at":"2025-02-19T18:59:30Z","abstract_excerpt":"Large language model (LLM) applications are evolving beyond simple chatbots into dynamic, general-purpose agentic programs, which scale LLM calls and output tokens to help AI agents reason, explore, and solve complex tasks. However, existing LLM serving systems ignore dependencies between programs and calls, missing significant opportunities for optimization. Our analysis reveals that programs submitted to LLM serving engines experience long cumulative wait times, primarily due to head-of-line blocking at both the individual LLM request and the program. To address this, we introduce Autellix, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.13965","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-19T18:59:30Z","cross_cats_sorted":["cs.AI","cs.DC"],"title_canon_sha256":"c3cf055d99103243be9a977f1fb88e1e8c6fe22ad90822e832732f9035c5a4a1","abstract_canon_sha256":"5495b78624661e9cdbd059748c48e0edaf00a7cf5138dc6e8a8264742aa32c53"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:03.186293Z","signature_b64":"X24jH1b5TGhx8DAHRKNTpsWKih8UZuW51LcMsfQAfncl9TMPfaRifnWZcqJ3DlVrIVTJwVnKFGFHjtiHaEWrBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"559e157860daebdc3ad3ed6fd2b4ccc405c07399009606c8c8416fe92c4cd5b6","last_reissued_at":"2026-07-05T10:17:03.185872Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:03.185872Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Autellix: An Efficient Serving Engine for LLM Agents as General Programs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DC"],"primary_cat":"cs.LG","authors_text":"Chi Wang, Colin Cai, Ion Stoica, Joseph E. Gonzalez, Justin Wong, Michael Luo, Tianjun Zhang, Xiaoxiang Shi, Yanping Huang, Yichuan Wang, Zhifeng Chen","submitted_at":"2025-02-19T18:59:30Z","abstract_excerpt":"Large language model (LLM) applications are evolving beyond simple chatbots into dynamic, general-purpose agentic programs, which scale LLM calls and output tokens to help AI agents reason, explore, and solve complex tasks. However, existing LLM serving systems ignore dependencies between programs and calls, missing significant opportunities for optimization. Our analysis reveals that programs submitted to LLM serving engines experience long cumulative wait times, primarily due to head-of-line blocking at both the individual LLM request and the program. To address this, we introduce Autellix, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13965","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13965/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.13965","created_at":"2026-07-05T10:17:03.185927+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.13965v1","created_at":"2026-07-05T10:17:03.185927+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13965","created_at":"2026-07-05T10:17:03.185927+00:00"},{"alias_kind":"pith_short_12","alias_value":"KWPBK6DA3LV5","created_at":"2026-07-05T10:17:03.185927+00:00"},{"alias_kind":"pith_short_16","alias_value":"KWPBK6DA3LV5YOWT","created_at":"2026-07-05T10:17:03.185927+00:00"},{"alias_kind":"pith_short_8","alias_value":"KWPBK6DA","created_at":"2026-07-05T10:17:03.185927+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08565","citing_title":"SMetric: Rethink LLM Scheduling for Serving Agents with Balanced Session-centric Scheduling","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18431","citing_title":"Beyond Prediction: Tail-Aware Scheduling for LLM Inference","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09613","citing_title":"AGENTSERVESIM: A Hardware-aware Simulator for Multi-Turn LLM Agent Serving","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00151","citing_title":"SmoothAgent: Efficient Long-Horizon LLM-Based Agent Serving with Lookahead Context Engineering","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02964","citing_title":"Multi-Segment Attention: Enabling Efficient KV-Cache Management for Faster Large Language Model Serving","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00866","citing_title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00528","citing_title":"SAGA: Workflow-Atomic Scheduling for AI Agent Inference on GPU Clusters","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27744","citing_title":"A Policy-Driven Runtime Layer for Agentic LLM Serving","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18586","citing_title":"TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02230","citing_title":"Continuum: Efficient and Robust Multi-Turn LLM Agent Scheduling with KV Cache Time-to-Live","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09611","citing_title":"Characterizing Performance-Energy Trade-offs of Large Language Models in Multi-Request Workflows","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03143","citing_title":"TokenDance: Scaling Multi-Agent LLM Serving via Collective KV Cache Sharing","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11030","citing_title":"An Executable Benchmarking Suite for Tool-Using Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06068","citing_title":"VibeServe: Can AI Agents Build Bespoke LLM Serving Systems?","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06472","citing_title":"Efficient Serving for Dynamic Agent Workflows with Prediction-based KV-Cache Management","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00528","citing_title":"SAGA: Workflow-Atomic Scheduling for AI Agent Inference on GPU Clusters","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10180","citing_title":"Tessera: Unlocking Heterogeneous GPUs through Kernel-Granularity Disaggregation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07238","citing_title":"FATE: Future-State-Aware Scheduling for Heterogeneous LLM Workflows","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06296","citing_title":"AgentOpt v0.1 Technical Report: Client-Side Optimization for LLM-Based Agent","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06370","citing_title":"ForkKV: Scaling Multi-LoRA Agent Serving via Copy-on-Write Disaggregated KV Cache","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26963","citing_title":"MARS: Efficient, Adaptive Co-Scheduling for Heterogeneous Agentic Systems","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15186","citing_title":"Scepsy: Serving Agentic Workflows Using Aggregate LLM Pipelines","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16682","citing_title":"KAIROS: Stateful, Context-Aware Power-Efficient Agentic Inference Serving","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ","json":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ.json","graph_json":"https://pith.science/api/pith-number/KWPBK6DA3LV5YOWT5VX5FNGMYQ/graph.json","events_json":"https://pith.science/api/pith-number/KWPBK6DA3LV5YOWT5VX5FNGMYQ/events.json","paper":"https://pith.science/paper/KWPBK6DA"},"agent_actions":{"view_html":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ","download_json":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ.json","view_paper":"https://pith.science/paper/KWPBK6DA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.13965&json=true","fetch_graph":"https://pith.science/api/pith-number/KWPBK6DA3LV5YOWT5VX5FNGMYQ/graph.json","fetch_events":"https://pith.science/api/pith-number/KWPBK6DA3LV5YOWT5VX5FNGMYQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ/action/storage_attestation","attest_author":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ/action/author_attestation","sign_citation":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ/action/citation_signature","submit_replication":"https://pith.science/pith/KWPBK6DA3LV5YOWT5VX5FNGMYQ/action/replication_record"}},"created_at":"2026-07-05T10:17:03.185927+00:00","updated_at":"2026-07-05T10:17:03.185927+00:00"}