{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4453U35VPK6BFJNFBMPE2N6IMG","short_pith_number":"pith:4453U35V","schema_version":"1.0","canonical_sha256":"e73bba6fb57abc12a5a50b1e4d37c861a89c72bf212ec775af6c3b635f5ad12b","source":{"kind":"arxiv","id":"2504.19867","version":1},"attestation_state":"computed","paper":{"title":"semi-PD: Towards Efficient LLM Serving via Phase-Wise Disaggregated Computation and Unified Storage","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.DC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Buhe Han, Chao Xiong, Guanyu Wu, Guohao Dai, Jianping Ma, Ke Hong, Lufang Chen, Qiuli Mao, Xiuhong Li, Yun Liang, Yu Wang, Zhong Wang","submitted_at":"2025-04-28T15:00:03Z","abstract_excerpt":"Existing large language model (LLM) serving systems fall into two categories: 1) a unified system where prefill phase and decode phase are co-located on the same GPU, sharing the unified computational resource and storage, and 2) a disaggregated system where the two phases are disaggregated to different GPUs. The design of the disaggregated system addresses the latency interference and sophisticated scheduling issues in the unified system but leads to storage challenges including 1) replicated weights for both phases that prevent flexible deployment, 2) KV cache transfer overhead between the t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.19867","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-28T15:00:03Z","cross_cats_sorted":["cs.DC","cs.LG"],"title_canon_sha256":"d643e58af47f245cd0b2001f4b6ab586d918c7e7e509ea293cdf4f4839f26278","abstract_canon_sha256":"2628b8b9d2b1d26557f15008821a0310c65112ed34bf7d8ec3f5948d5562d050"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:04.940639Z","signature_b64":"x5UUE96QFkTUudia5xro+SGLuYsSzBZESWfwHdVN7h8m3eXC5hRF8LCkD4uXgVNlHTMzToNslL0/I0F5XvQaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e73bba6fb57abc12a5a50b1e4d37c861a89c72bf212ec775af6c3b635f5ad12b","last_reissued_at":"2026-07-05T10:55:04.940162Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:04.940162Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"semi-PD: Towards Efficient LLM Serving via Phase-Wise Disaggregated Computation and Unified Storage","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.DC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Buhe Han, Chao Xiong, Guanyu Wu, Guohao Dai, Jianping Ma, Ke Hong, Lufang Chen, Qiuli Mao, Xiuhong Li, Yun Liang, Yu Wang, Zhong Wang","submitted_at":"2025-04-28T15:00:03Z","abstract_excerpt":"Existing large language model (LLM) serving systems fall into two categories: 1) a unified system where prefill phase and decode phase are co-located on the same GPU, sharing the unified computational resource and storage, and 2) a disaggregated system where the two phases are disaggregated to different GPUs. The design of the disaggregated system addresses the latency interference and sophisticated scheduling issues in the unified system but leads to storage challenges including 1) replicated weights for both phases that prevent flexible deployment, 2) KV cache transfer overhead between the t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.19867","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.19867/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.19867","created_at":"2026-07-05T10:55:04.940219+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.19867v1","created_at":"2026-07-05T10:55:04.940219+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.19867","created_at":"2026-07-05T10:55:04.940219+00:00"},{"alias_kind":"pith_short_12","alias_value":"4453U35VPK6B","created_at":"2026-07-05T10:55:04.940219+00:00"},{"alias_kind":"pith_short_16","alias_value":"4453U35VPK6BFJNF","created_at":"2026-07-05T10:55:04.940219+00:00"},{"alias_kind":"pith_short_8","alias_value":"4453U35V","created_at":"2026-07-05T10:55:04.940219+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10537","citing_title":"Prefilling-dLLM: Predictive Prefilling for Long-Context Inference in Diffusion Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04415","citing_title":"FlexNPU: Transparent NPU Virtualization for Dynamic LLM Prefill-Decode Co-location","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13734","citing_title":"KVServe: Service-Aware KV Cache Compression for Communication-Efficient Disaggregated LLM Serving","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG","json":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG.json","graph_json":"https://pith.science/api/pith-number/4453U35VPK6BFJNFBMPE2N6IMG/graph.json","events_json":"https://pith.science/api/pith-number/4453U35VPK6BFJNFBMPE2N6IMG/events.json","paper":"https://pith.science/paper/4453U35V"},"agent_actions":{"view_html":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG","download_json":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG.json","view_paper":"https://pith.science/paper/4453U35V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.19867&json=true","fetch_graph":"https://pith.science/api/pith-number/4453U35VPK6BFJNFBMPE2N6IMG/graph.json","fetch_events":"https://pith.science/api/pith-number/4453U35VPK6BFJNFBMPE2N6IMG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG/action/storage_attestation","attest_author":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG/action/author_attestation","sign_citation":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG/action/citation_signature","submit_replication":"https://pith.science/pith/4453U35VPK6BFJNFBMPE2N6IMG/action/replication_record"}},"created_at":"2026-07-05T10:55:04.940219+00:00","updated_at":"2026-07-05T10:55:04.940219+00:00"}