{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WRXJHIX6U2ZPJZOGOKBJLHJUT5","short_pith_number":"pith:WRXJHIX6","schema_version":"1.0","canonical_sha256":"b46e93a2fea6b2f4e5c67282959d349f5615a37e20c1506a3f820f4df94795cd","source":{"kind":"arxiv","id":"2405.01814","version":2},"attestation_state":"computed","paper":{"title":"Efficient Heterogeneous Large Language Model Decoding with Model-Attention Disaggregation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Jinlei Jiang, Kang Chen, Mingxing Zhang, Shaoyuan Chen, Wencong Xiao, Yingdi Shan, Yongwei Wu, Yutong Lin","submitted_at":"2024-05-03T02:15:15Z","abstract_excerpt":"Transformer-based large language models (LLMs) exhibit impressive performance in generative tasks but also introduce significant challenges in real-world serving due to inefficient use of the expensive, computation-optimized accelerators. Although disaggregated serving architectures have been proposed to split different phases of LLM inference, the efficiency of decoding phase is still low. This is caused by the varying resource demands of different operators in the transformer-based LLMs. Specifically, the attention operator is memory-intensive, exhibiting a memory access pattern that clashes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.01814","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-03T02:15:15Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"02c107d39c976bf8aff1d4c7cd68bd23c10175b51fd74af6e179c859eb3ea3f1","abstract_canon_sha256":"f7c740433e2708b0d7785000c25e019ffc498cfccc05b32db178b6c7b6536d92"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:54.069507Z","signature_b64":"Syp5IwvKyZAt+atpu0GvBIeX+6iJJiuctP6NpMWzKIu6z/iEJxFYatRa5fFMhpFbkkieuFoK/wwfdP8cwBWMAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b46e93a2fea6b2f4e5c67282959d349f5615a37e20c1506a3f820f4df94795cd","last_reissued_at":"2026-07-05T10:46:54.069029Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:54.069029Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Heterogeneous Large Language Model Decoding with Model-Attention Disaggregation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Jinlei Jiang, Kang Chen, Mingxing Zhang, Shaoyuan Chen, Wencong Xiao, Yingdi Shan, Yongwei Wu, Yutong Lin","submitted_at":"2024-05-03T02:15:15Z","abstract_excerpt":"Transformer-based large language models (LLMs) exhibit impressive performance in generative tasks but also introduce significant challenges in real-world serving due to inefficient use of the expensive, computation-optimized accelerators. Although disaggregated serving architectures have been proposed to split different phases of LLM inference, the efficiency of decoding phase is still low. This is caused by the varying resource demands of different operators in the transformer-based LLMs. Specifically, the attention operator is memory-intensive, exhibiting a memory access pattern that clashes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01814","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.01814/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.01814","created_at":"2026-07-05T10:46:54.069089+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.01814v2","created_at":"2026-07-05T10:46:54.069089+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01814","created_at":"2026-07-05T10:46:54.069089+00:00"},{"alias_kind":"pith_short_12","alias_value":"WRXJHIX6U2ZP","created_at":"2026-07-05T10:46:54.069089+00:00"},{"alias_kind":"pith_short_16","alias_value":"WRXJHIX6U2ZPJZOG","created_at":"2026-07-05T10:46:54.069089+00:00"},{"alias_kind":"pith_short_8","alias_value":"WRXJHIX6","created_at":"2026-07-05T10:46:54.069089+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.10726","citing_title":"PrefixWall: Mitigating Prefix Caching Side Channels in Shared LLM Systems","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2409.10516","citing_title":"RetrievalAttention: Accelerating Long-Context LLM Inference via Vector Retrieval","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13525","citing_title":"Janus: Disaggregating Attention and Experts for Scalable MoE Inference","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5","json":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5.json","graph_json":"https://pith.science/api/pith-number/WRXJHIX6U2ZPJZOGOKBJLHJUT5/graph.json","events_json":"https://pith.science/api/pith-number/WRXJHIX6U2ZPJZOGOKBJLHJUT5/events.json","paper":"https://pith.science/paper/WRXJHIX6"},"agent_actions":{"view_html":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5","download_json":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5.json","view_paper":"https://pith.science/paper/WRXJHIX6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.01814&json=true","fetch_graph":"https://pith.science/api/pith-number/WRXJHIX6U2ZPJZOGOKBJLHJUT5/graph.json","fetch_events":"https://pith.science/api/pith-number/WRXJHIX6U2ZPJZOGOKBJLHJUT5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5/action/storage_attestation","attest_author":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5/action/author_attestation","sign_citation":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5/action/citation_signature","submit_replication":"https://pith.science/pith/WRXJHIX6U2ZPJZOGOKBJLHJUT5/action/replication_record"}},"created_at":"2026-07-05T10:46:54.069089+00:00","updated_at":"2026-07-05T10:46:54.069089+00:00"}