{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QLZYHMPRBUYIXT4QDQW2NGE3CF","short_pith_number":"pith:QLZYHMPR","schema_version":"1.0","canonical_sha256":"82f383b1f10d308bcf901c2da6989b117feb369af692a007ce9c094f63f536d5","source":{"kind":"arxiv","id":"2508.15881","version":2},"attestation_state":"computed","paper":{"title":"TPLA: Tensor Parallel Latent Attention for Efficient Disaggregated Prefill and Decode Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Di Yin, Fanxu Meng, Muhan Zhang, Pingzhi Tang, Xiaojuan Tang, Xing Sun, Yuxuan Wang","submitted_at":"2025-08-21T15:25:40Z","abstract_excerpt":"Multi-Head Latent Attention (MLA), introduced in DeepSeek-V2, compresses key-value states into a low-rank latent vector, caching only this vector to reduce memory. In tensor parallelism (TP), however, attention heads are computed across multiple devices, and each device must load the full cache, eroding the advantage of MLA over Grouped Query Attention (GQA). We propose Tensor-Parallel Latent Attention (TPLA): a scheme that partitions both the latent representation and each head's input dimension across devices, performs attention independently per shard, and then combines results with an all-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.15881","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-21T15:25:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f63a4bbfd421128841717a214bc3806a1d7497b4b908ec0d36bfeff3a35a9c83","abstract_canon_sha256":"831435e28ecb711f59c6832dfcd4019015bfddef6dcb830b71e883b8b07a55a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:36.411896Z","signature_b64":"9YcIhCPlnpZfn5G4WEP1PRbb1io+4YYi73BEb2EmfMs/YpGHbLublabm08hCc0RCzy1eCYQbA9/UzuN5Mbu4CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"82f383b1f10d308bcf901c2da6989b117feb369af692a007ce9c094f63f536d5","last_reissued_at":"2026-07-05T11:58:36.411424Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:36.411424Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TPLA: Tensor Parallel Latent Attention for Efficient Disaggregated Prefill and Decode Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Di Yin, Fanxu Meng, Muhan Zhang, Pingzhi Tang, Xiaojuan Tang, Xing Sun, Yuxuan Wang","submitted_at":"2025-08-21T15:25:40Z","abstract_excerpt":"Multi-Head Latent Attention (MLA), introduced in DeepSeek-V2, compresses key-value states into a low-rank latent vector, caching only this vector to reduce memory. In tensor parallelism (TP), however, attention heads are computed across multiple devices, and each device must load the full cache, eroding the advantage of MLA over Grouped Query Attention (GQA). We propose Tensor-Parallel Latent Attention (TPLA): a scheme that partitions both the latent representation and each head's input dimension across devices, performs attention independently per shard, and then combines results with an all-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.15881","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.15881/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.15881","created_at":"2026-07-05T11:58:36.411482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.15881v2","created_at":"2026-07-05T11:58:36.411482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.15881","created_at":"2026-07-05T11:58:36.411482+00:00"},{"alias_kind":"pith_short_12","alias_value":"QLZYHMPRBUYI","created_at":"2026-07-05T11:58:36.411482+00:00"},{"alias_kind":"pith_short_16","alias_value":"QLZYHMPRBUYIXT4Q","created_at":"2026-07-05T11:58:36.411482+00:00"},{"alias_kind":"pith_short_8","alias_value":"QLZYHMPR","created_at":"2026-07-05T11:58:36.411482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05876","citing_title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05876","citing_title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF","json":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF.json","graph_json":"https://pith.science/api/pith-number/QLZYHMPRBUYIXT4QDQW2NGE3CF/graph.json","events_json":"https://pith.science/api/pith-number/QLZYHMPRBUYIXT4QDQW2NGE3CF/events.json","paper":"https://pith.science/paper/QLZYHMPR"},"agent_actions":{"view_html":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF","download_json":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF.json","view_paper":"https://pith.science/paper/QLZYHMPR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.15881&json=true","fetch_graph":"https://pith.science/api/pith-number/QLZYHMPRBUYIXT4QDQW2NGE3CF/graph.json","fetch_events":"https://pith.science/api/pith-number/QLZYHMPRBUYIXT4QDQW2NGE3CF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF/action/storage_attestation","attest_author":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF/action/author_attestation","sign_citation":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF/action/citation_signature","submit_replication":"https://pith.science/pith/QLZYHMPRBUYIXT4QDQW2NGE3CF/action/replication_record"}},"created_at":"2026-07-05T11:58:36.411482+00:00","updated_at":"2026-07-05T11:58:36.411482+00:00"}