{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CSM4UHHKBRHZQ23KOI4VWH35YF","short_pith_number":"pith:CSM4UHHK","schema_version":"1.0","canonical_sha256":"1499ca1cea0c4f986b6a72395b1f7dc14ed003ca77849afe409853fc0762f2fb","source":{"kind":"arxiv","id":"2412.04964","version":2},"attestation_state":"computed","paper":{"title":"Flash Communication: Reducing Tensor Parallelization Bottleneck for Fast Large Language Model Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bo Zhang, Liang Ye, Lin Ma, Qingyuan Li, Wei Wu, Yerui Sun, Yifan Zhang, Yuchen Xie","submitted_at":"2024-12-06T11:29:32Z","abstract_excerpt":"The ever-increasing sizes of large language models necessitate distributed solutions for fast inference that exploit multi-dimensional parallelism, where computational loads are split across various accelerators such as GPU clusters. However, this approach often introduces significant communication overhead, especially on devices with limited bandwidth. In this paper, we introduce Flash Communication, a novel low-bit compression technique designed to alleviate the tensor-parallelism communication bottleneck during inference. Our method substantially boosts intra-node communication speed by mor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.04964","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-12-06T11:29:32Z","cross_cats_sorted":[],"title_canon_sha256":"41bcf5bcb3f49564c2ef02be422ae444d9dc1848c72f65a2b283ce9c20ccad0d","abstract_canon_sha256":"72a86d13f329f302a99505d4a48b807c3c9391eca7d21113f01e3d35449a4001"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:37.499972Z","signature_b64":"Nfg/ntaSiF2ZpY5dRAhGv6mmP+ibNHTz2JCpqQI+OEuKA1sbnKToS+r5Sw1Orxz1bAUzCDbAFSVWYB1zfRd9AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1499ca1cea0c4f986b6a72395b1f7dc14ed003ca77849afe409853fc0762f2fb","last_reissued_at":"2026-07-05T09:47:37.499488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:37.499488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Flash Communication: Reducing Tensor Parallelization Bottleneck for Fast Large Language Model Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Bo Zhang, Liang Ye, Lin Ma, Qingyuan Li, Wei Wu, Yerui Sun, Yifan Zhang, Yuchen Xie","submitted_at":"2024-12-06T11:29:32Z","abstract_excerpt":"The ever-increasing sizes of large language models necessitate distributed solutions for fast inference that exploit multi-dimensional parallelism, where computational loads are split across various accelerators such as GPU clusters. However, this approach often introduces significant communication overhead, especially on devices with limited bandwidth. In this paper, we introduce Flash Communication, a novel low-bit compression technique designed to alleviate the tensor-parallelism communication bottleneck during inference. Our method substantially boosts intra-node communication speed by mor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.04964","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.04964/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.04964","created_at":"2026-07-05T09:47:37.499547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.04964v2","created_at":"2026-07-05T09:47:37.499547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.04964","created_at":"2026-07-05T09:47:37.499547+00:00"},{"alias_kind":"pith_short_12","alias_value":"CSM4UHHKBRHZ","created_at":"2026-07-05T09:47:37.499547+00:00"},{"alias_kind":"pith_short_16","alias_value":"CSM4UHHKBRHZQ23K","created_at":"2026-07-05T09:47:37.499547+00:00"},{"alias_kind":"pith_short_8","alias_value":"CSM4UHHK","created_at":"2026-07-05T09:47:37.499547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.21351","citing_title":"Analytical Provisioning for Attention-FFN Disaggregated LLM Serving under Stochastic Workloads","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28239","citing_title":"A Switch-Centric In-Network Architecture for Accelerating LLM Inference in Shared-Memory Network","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24088","citing_title":"TACO: Efficient Communication Compression of Intermediate Tensors for Scalable Tensor-Parallel LLM Training","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24013","citing_title":"CommFuse: Hiding Tail Latency via Communication Decomposition and Fusion for Distributed LLM Training","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF","json":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF.json","graph_json":"https://pith.science/api/pith-number/CSM4UHHKBRHZQ23KOI4VWH35YF/graph.json","events_json":"https://pith.science/api/pith-number/CSM4UHHKBRHZQ23KOI4VWH35YF/events.json","paper":"https://pith.science/paper/CSM4UHHK"},"agent_actions":{"view_html":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF","download_json":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF.json","view_paper":"https://pith.science/paper/CSM4UHHK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.04964&json=true","fetch_graph":"https://pith.science/api/pith-number/CSM4UHHKBRHZQ23KOI4VWH35YF/graph.json","fetch_events":"https://pith.science/api/pith-number/CSM4UHHKBRHZQ23KOI4VWH35YF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF/action/storage_attestation","attest_author":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF/action/author_attestation","sign_citation":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF/action/citation_signature","submit_replication":"https://pith.science/pith/CSM4UHHKBRHZQ23KOI4VWH35YF/action/replication_record"}},"created_at":"2026-07-05T09:47:37.499547+00:00","updated_at":"2026-07-05T09:47:37.499547+00:00"}