{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NAPOCMRPXNQXJNXBVY2FV2R4ES","short_pith_number":"pith:NAPOCMRP","schema_version":"1.0","canonical_sha256":"681ee1322fbb6174b6e1ae345aea3c24a3d6bfed38fd028f6a4a0552fd9ea031","source":{"kind":"arxiv","id":"2501.14417","version":3},"attestation_state":"computed","paper":{"title":"DeepServe: Serverless Large Language Model Serving at Scale","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Baoquan Zhang, Changhong Liu, Dayun Lin, Gengyuan Dan, Hao Feng, Hao Xu, Jiang Liu, Jiang Xu, Jie Meng, Junhao Hu, Qin Zhang, Shining Wan, Tao Xie, Xusheng Chen, Yizhou Shan, Yuetao Chen, Yue Yu, Yulong He, Zhihao Ren, Zhixia Liu, Zhiyu Dong","submitted_at":"2025-01-24T11:34:13Z","abstract_excerpt":"In this paper, we propose DEEPSERVE, a scalable and serverless AI platform designed to efficiently serve large language models (LLMs) at scale in cloud environments. DEEPSERVE addresses key challenges such as resource allocation, serving efficiency, and cold start latencies through four main design components. First, DEEPSERVE uses a simple serverless abstraction called the request-job-task model, which helps manage diverse AI workloads across posttraining and model-serving tasks. Second, DEEPSERVE integrates an in-house serving engine named FLOWSERVE using a microkernel-inspired design, NPU-c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14417","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-01-24T11:34:13Z","cross_cats_sorted":[],"title_canon_sha256":"2146ee6a129048012a549af1458a9b14052d553ab1f5a7210ab61ee2a4f24c85","abstract_canon_sha256":"78344ef5195b038710869d82ec68bd9424b96b60346032471cff008284634398"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:17.322670Z","signature_b64":"hQv6uSFtFfXf8ysFlvD3oG80YgL+4ZMc+Ryn1/J6dvvdLluYgZugy8h0OEDOv9KNejvTPhOQM0gwv8+WIzawDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"681ee1322fbb6174b6e1ae345aea3c24a3d6bfed38fd028f6a4a0552fd9ea031","last_reissued_at":"2026-07-05T11:18:17.322136Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:17.322136Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeepServe: Serverless Large Language Model Serving at Scale","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Baoquan Zhang, Changhong Liu, Dayun Lin, Gengyuan Dan, Hao Feng, Hao Xu, Jiang Liu, Jiang Xu, Jie Meng, Junhao Hu, Qin Zhang, Shining Wan, Tao Xie, Xusheng Chen, Yizhou Shan, Yuetao Chen, Yue Yu, Yulong He, Zhihao Ren, Zhixia Liu, Zhiyu Dong","submitted_at":"2025-01-24T11:34:13Z","abstract_excerpt":"In this paper, we propose DEEPSERVE, a scalable and serverless AI platform designed to efficiently serve large language models (LLMs) at scale in cloud environments. DEEPSERVE addresses key challenges such as resource allocation, serving efficiency, and cold start latencies through four main design components. First, DEEPSERVE uses a simple serverless abstraction called the request-job-task model, which helps manage diverse AI workloads across posttraining and model-serving tasks. Second, DEEPSERVE integrates an in-house serving engine named FLOWSERVE using a microkernel-inspired design, NPU-c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14417","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14417/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14417","created_at":"2026-07-05T11:18:17.322194+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14417v3","created_at":"2026-07-05T11:18:17.322194+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14417","created_at":"2026-07-05T11:18:17.322194+00:00"},{"alias_kind":"pith_short_12","alias_value":"NAPOCMRPXNQX","created_at":"2026-07-05T11:18:17.322194+00:00"},{"alias_kind":"pith_short_16","alias_value":"NAPOCMRPXNQXJNXB","created_at":"2026-07-05T11:18:17.322194+00:00"},{"alias_kind":"pith_short_8","alias_value":"NAPOCMRP","created_at":"2026-07-05T11:18:17.322194+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES","json":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES.json","graph_json":"https://pith.science/api/pith-number/NAPOCMRPXNQXJNXBVY2FV2R4ES/graph.json","events_json":"https://pith.science/api/pith-number/NAPOCMRPXNQXJNXBVY2FV2R4ES/events.json","paper":"https://pith.science/paper/NAPOCMRP"},"agent_actions":{"view_html":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES","download_json":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES.json","view_paper":"https://pith.science/paper/NAPOCMRP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14417&json=true","fetch_graph":"https://pith.science/api/pith-number/NAPOCMRPXNQXJNXBVY2FV2R4ES/graph.json","fetch_events":"https://pith.science/api/pith-number/NAPOCMRPXNQXJNXBVY2FV2R4ES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES/action/storage_attestation","attest_author":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES/action/author_attestation","sign_citation":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES/action/citation_signature","submit_replication":"https://pith.science/pith/NAPOCMRPXNQXJNXBVY2FV2R4ES/action/replication_record"}},"created_at":"2026-07-05T11:18:17.322194+00:00","updated_at":"2026-07-05T11:18:17.322194+00:00"}