{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ONG4BZ3G3YVB7NLHXOYHHQGBG3","short_pith_number":"pith:ONG4BZ3G","schema_version":"1.0","canonical_sha256":"734dc0e766de2a1fb567bbb073c0c136d05f8a61991c199a9dc49ac4219ecc6f","source":{"kind":"arxiv","id":"2504.03648","version":1},"attestation_state":"computed","paper":{"title":"AIBrix: Towards Scalable, Cost-Effective Large Language Model Inference Infrastructure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Binbin Chen, Caixue Lin, Chak-Pong Chung, Chenyu Jiang, Gangmuk Lim, Haiyang Shi, Jianjun Chen, Jingyuan Zhang, Kante Yin, Le Xu, Liguang Xie, Linhui Xu, Ning Wang, Rong Kang, Rui Shi, Shuowei Jin, The AIBrix Team: Jiaxin Shan, Tongping Liu, Varun Gupta, Wu Xiang, Xiao Liu, Xin Chen, Yicheng Lu, Yifei Zhang, Yiqing Zhu, Zuzhi Chen","submitted_at":"2025-02-22T07:07:38Z","abstract_excerpt":"We introduce AIBrix, a cloud-native, open-source framework designed to optimize and simplify large-scale LLM deployment in cloud environments. Unlike traditional cloud-native stacks, AIBrix follows a co-design philosophy, ensuring every layer of the infrastructure is purpose-built for seamless integration with inference engines like vLLM. AIBrix introduces several key innovations to reduce inference costs and enhance performance including high-density LoRA management for dynamic adapter scheduling, LLM-specific autoscalers, and prefix-aware, load-aware routing. To further improve efficiency, A"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.03648","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-02-22T07:07:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b40d8a37dda302b8340b46cdc293131f3d1bb6d0d01c61a95bef830a19e4bc37","abstract_canon_sha256":"ae0701014c47cb143ecac141ecca01328c79e48381240d7d6ca9d232f4363ee8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:45.507066Z","signature_b64":"MJKxbczPTVHQu9ZJ4K0++qlju+Az77xORHHX/7Bf5BpfSu9BNxsVP1QqH399tLn3k4OJhXFJchYXGExageqfCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"734dc0e766de2a1fb567bbb073c0c136d05f8a61991c199a9dc49ac4219ecc6f","last_reissued_at":"2026-07-05T10:44:45.506573Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:45.506573Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AIBrix: Towards Scalable, Cost-Effective Large Language Model Inference Infrastructure","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Binbin Chen, Caixue Lin, Chak-Pong Chung, Chenyu Jiang, Gangmuk Lim, Haiyang Shi, Jianjun Chen, Jingyuan Zhang, Kante Yin, Le Xu, Liguang Xie, Linhui Xu, Ning Wang, Rong Kang, Rui Shi, Shuowei Jin, The AIBrix Team: Jiaxin Shan, Tongping Liu, Varun Gupta, Wu Xiang, Xiao Liu, Xin Chen, Yicheng Lu, Yifei Zhang, Yiqing Zhu, Zuzhi Chen","submitted_at":"2025-02-22T07:07:38Z","abstract_excerpt":"We introduce AIBrix, a cloud-native, open-source framework designed to optimize and simplify large-scale LLM deployment in cloud environments. Unlike traditional cloud-native stacks, AIBrix follows a co-design philosophy, ensuring every layer of the infrastructure is purpose-built for seamless integration with inference engines like vLLM. AIBrix introduces several key innovations to reduce inference costs and enhance performance including high-density LoRA management for dynamic adapter scheduling, LLM-specific autoscalers, and prefix-aware, load-aware routing. To further improve efficiency, A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.03648","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.03648/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.03648","created_at":"2026-07-05T10:44:45.506631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.03648v1","created_at":"2026-07-05T10:44:45.506631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.03648","created_at":"2026-07-05T10:44:45.506631+00:00"},{"alias_kind":"pith_short_12","alias_value":"ONG4BZ3G3YVB","created_at":"2026-07-05T10:44:45.506631+00:00"},{"alias_kind":"pith_short_16","alias_value":"ONG4BZ3G3YVB7NLH","created_at":"2026-07-05T10:44:45.506631+00:00"},{"alias_kind":"pith_short_8","alias_value":"ONG4BZ3G","created_at":"2026-07-05T10:44:45.506631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07862","citing_title":"CTA-Pipelining: A Latency-Oriented Spatial Scaling Method for Multi-GPU Systems","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12950","citing_title":"Maestro: Workload-Aware Cross-Cluster Scheduling for LLM-Based Multi-Agent Systems","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19179","citing_title":"CascadeInfer: Length-Aware Scheduling of LLM Serving with Low Latency and Load Balancing","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16867","citing_title":"GoodServe: Towards High-Goodput Serving of Agentic LLM Inferences over Heterogeneous Resources","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20979","citing_title":"Toward Robust and Efficient ML-Based GPU Caching for Modern Inference","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07443","citing_title":"RcLLM: Accelerating Generative Recommendation via Beyond-Prefix KV Caching","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15186","citing_title":"Scepsy: Serving Agentic Workflows Using Aggregate LLM Pipelines","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3","json":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3.json","graph_json":"https://pith.science/api/pith-number/ONG4BZ3G3YVB7NLHXOYHHQGBG3/graph.json","events_json":"https://pith.science/api/pith-number/ONG4BZ3G3YVB7NLHXOYHHQGBG3/events.json","paper":"https://pith.science/paper/ONG4BZ3G"},"agent_actions":{"view_html":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3","download_json":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3.json","view_paper":"https://pith.science/paper/ONG4BZ3G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.03648&json=true","fetch_graph":"https://pith.science/api/pith-number/ONG4BZ3G3YVB7NLHXOYHHQGBG3/graph.json","fetch_events":"https://pith.science/api/pith-number/ONG4BZ3G3YVB7NLHXOYHHQGBG3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3/action/storage_attestation","attest_author":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3/action/author_attestation","sign_citation":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3/action/citation_signature","submit_replication":"https://pith.science/pith/ONG4BZ3G3YVB7NLHXOYHHQGBG3/action/replication_record"}},"created_at":"2026-07-05T10:44:45.506631+00:00","updated_at":"2026-07-05T10:44:45.506631+00:00"}