{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7ZJKKDJ3BHGNRXIATC3B2LLZYR","short_pith_number":"pith:7ZJKKDJ3","schema_version":"1.0","canonical_sha256":"fe52a50d3b09ccd8dd0098b61d2d79c4683b80a5ad0210d420d62a8fadb835ba","source":{"kind":"arxiv","id":"2501.05460","version":4},"attestation_state":"computed","paper":{"title":"Efficiently Serving Large Multimodal Models Using EPD Disaggregation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.DC","authors_text":"Gursimran Singh, Linzi Xing, Timothy Yu, Wei Jiang, Xiaolong Bai, Xinglu Wang, Yifan Hu, Yi Li, Ying Xiong, Yong Zhang, Zhefeng Wang, Zhenan Fan","submitted_at":"2024-12-25T10:11:31Z","abstract_excerpt":"Large Multimodal Models (LMMs) extend Large Language Models (LLMs) by handling diverse inputs such as images, audio, and video, but at the cost of adding a multimodal encoding stage that increases both computational and memory overhead. This step negatively affects key Service Level Objectives (SLOs), such as time to first token (TTFT) and time per output token (TPOT). We introduce Encode-Prefill-Decode (EPD) Disaggregation, a novel framework that separates the encoding, prefill, and decode stages onto dedicated resources. Unlike current systems, which bundle encoding and prefill together, our"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05460","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-12-25T10:11:31Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"72bc63e0f6779b17c4a4121501d80ba8cd57714c1f8ea3d5ec7adabf5cf17a0b","abstract_canon_sha256":"d80fed5f877d51d3160b088c8a8ded3b07fdd26c4efabaa9c8c307b5441e00d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:33.376191Z","signature_b64":"Vy1ctmbtAJtodB4+khR1pZtAnoiYlEVXrfXHnCF4/S18ICSdi1G7GAkuZnoHfmDkdqD6efDMQaOmceKwDBHbBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fe52a50d3b09ccd8dd0098b61d2d79c4683b80a5ad0210d420d62a8fadb835ba","last_reissued_at":"2026-07-05T11:28:33.375403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:33.375403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficiently Serving Large Multimodal Models Using EPD Disaggregation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.DC","authors_text":"Gursimran Singh, Linzi Xing, Timothy Yu, Wei Jiang, Xiaolong Bai, Xinglu Wang, Yifan Hu, Yi Li, Ying Xiong, Yong Zhang, Zhefeng Wang, Zhenan Fan","submitted_at":"2024-12-25T10:11:31Z","abstract_excerpt":"Large Multimodal Models (LMMs) extend Large Language Models (LLMs) by handling diverse inputs such as images, audio, and video, but at the cost of adding a multimodal encoding stage that increases both computational and memory overhead. This step negatively affects key Service Level Objectives (SLOs), such as time to first token (TTFT) and time per output token (TPOT). We introduce Encode-Prefill-Decode (EPD) Disaggregation, a novel framework that separates the encoding, prefill, and decode stages onto dedicated resources. Unlike current systems, which bundle encoding and prefill together, our"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05460","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05460","created_at":"2026-07-05T11:28:33.375675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05460v4","created_at":"2026-07-05T11:28:33.375675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05460","created_at":"2026-07-05T11:28:33.375675+00:00"},{"alias_kind":"pith_short_12","alias_value":"7ZJKKDJ3BHGN","created_at":"2026-07-05T11:28:33.375675+00:00"},{"alias_kind":"pith_short_16","alias_value":"7ZJKKDJ3BHGNRXIA","created_at":"2026-07-05T11:28:33.375675+00:00"},{"alias_kind":"pith_short_8","alias_value":"7ZJKKDJ3","created_at":"2026-07-05T11:28:33.375675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12688","citing_title":"M*: A Modular, Extensible, Serving System for Multimodal Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29639","citing_title":"RTP-LLM: High-Performance Alibaba LLM Inference Engine","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02329","citing_title":"Taming Request Imbalance: SLO-Aware Scheduling for Disaggregated LLM Inference","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR","json":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR.json","graph_json":"https://pith.science/api/pith-number/7ZJKKDJ3BHGNRXIATC3B2LLZYR/graph.json","events_json":"https://pith.science/api/pith-number/7ZJKKDJ3BHGNRXIATC3B2LLZYR/events.json","paper":"https://pith.science/paper/7ZJKKDJ3"},"agent_actions":{"view_html":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR","download_json":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR.json","view_paper":"https://pith.science/paper/7ZJKKDJ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05460&json=true","fetch_graph":"https://pith.science/api/pith-number/7ZJKKDJ3BHGNRXIATC3B2LLZYR/graph.json","fetch_events":"https://pith.science/api/pith-number/7ZJKKDJ3BHGNRXIATC3B2LLZYR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR/action/storage_attestation","attest_author":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR/action/author_attestation","sign_citation":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR/action/citation_signature","submit_replication":"https://pith.science/pith/7ZJKKDJ3BHGNRXIATC3B2LLZYR/action/replication_record"}},"created_at":"2026-07-05T11:28:33.375675+00:00","updated_at":"2026-07-05T11:28:33.375675+00:00"}