{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YY4UVKZWWBWBVLXPEV52QQPXYC","short_pith_number":"pith:YY4UVKZW","schema_version":"1.0","canonical_sha256":"c6394aab36b06c1aaeef257ba841f7c0aeb6f5161c086cd9efb1469c50dd8c2b","source":{"kind":"arxiv","id":"2404.19429","version":1},"attestation_state":"computed","paper":{"title":"Lancet: Accelerating Mixture-of-Experts Training via Whole Graph Computation-Communication Overlapping","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Chenyu Jiang, Chuan Wu, Shuai Zheng, Ye Tian, Yida Wang, Zhen Jia","submitted_at":"2024-04-30T10:17:21Z","abstract_excerpt":"The Mixture-of-Expert (MoE) technique plays a crucial role in expanding the size of DNN model parameters. However, it faces the challenge of extended all-to-all communication latency during the training process. Existing methods attempt to mitigate this issue by overlapping all-to-all with expert computation. Yet, these methods frequently fall short of achieving sufficient overlap, consequently restricting the potential for performance enhancements. In our study, we extend the scope of this challenge by considering overlap at the broader training graph level. During the forward pass, we enable"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.19429","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-04-30T10:17:21Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"4c6a7ba1498591e9fb1fa069262ae2a29e52277f5c8fa051be1b2c88f7a1914a","abstract_canon_sha256":"6c4a37525e4e5e2942d8c32636547841a3579f5d278af0f08ece275ccfe64036"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:13:43.650475Z","signature_b64":"0cxoDmicpS+ZPf3xkfFyEwwr4vA0RT6YXR6htW6D0vL4DFnXcuxEzvnWBjCzCULRSbn9H0pHtNWy+6FdmSzXBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c6394aab36b06c1aaeef257ba841f7c0aeb6f5161c086cd9efb1469c50dd8c2b","last_reissued_at":"2026-07-05T08:13:43.649942Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:13:43.649942Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lancet: Accelerating Mixture-of-Experts Training via Whole Graph Computation-Communication Overlapping","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Chenyu Jiang, Chuan Wu, Shuai Zheng, Ye Tian, Yida Wang, Zhen Jia","submitted_at":"2024-04-30T10:17:21Z","abstract_excerpt":"The Mixture-of-Expert (MoE) technique plays a crucial role in expanding the size of DNN model parameters. However, it faces the challenge of extended all-to-all communication latency during the training process. Existing methods attempt to mitigate this issue by overlapping all-to-all with expert computation. Yet, these methods frequently fall short of achieving sufficient overlap, consequently restricting the potential for performance enhancements. In our study, we extend the scope of this challenge by considering overlap at the broader training graph level. During the forward pass, we enable"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.19429","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.19429/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.19429","created_at":"2026-07-05T08:13:43.650008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.19429v1","created_at":"2026-07-05T08:13:43.650008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.19429","created_at":"2026-07-05T08:13:43.650008+00:00"},{"alias_kind":"pith_short_12","alias_value":"YY4UVKZWWBWB","created_at":"2026-07-05T08:13:43.650008+00:00"},{"alias_kind":"pith_short_16","alias_value":"YY4UVKZWWBWBVLXP","created_at":"2026-07-05T08:13:43.650008+00:00"},{"alias_kind":"pith_short_8","alias_value":"YY4UVKZW","created_at":"2026-07-05T08:13:43.650008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09200","citing_title":"Resource-aware Computation-Communication Overlap for multi-GPU ML Workloads","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23764","citing_title":"HyperParallel-MoE: Multi-Core Interleaved Scheduling for Fast MoE Training on Ascend NPUs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28516","citing_title":"CLEAR-MoE: Shared-Basis Expert Extraction from Frozen Vision Transformers via Calibration-Driven Layer Selection","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23764","citing_title":"HyperParallel-MoE: Multi-Core Interleaved Scheduling for Fast MoE Training on Ascend NPUs","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC","json":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC.json","graph_json":"https://pith.science/api/pith-number/YY4UVKZWWBWBVLXPEV52QQPXYC/graph.json","events_json":"https://pith.science/api/pith-number/YY4UVKZWWBWBVLXPEV52QQPXYC/events.json","paper":"https://pith.science/paper/YY4UVKZW"},"agent_actions":{"view_html":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC","download_json":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC.json","view_paper":"https://pith.science/paper/YY4UVKZW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.19429&json=true","fetch_graph":"https://pith.science/api/pith-number/YY4UVKZWWBWBVLXPEV52QQPXYC/graph.json","fetch_events":"https://pith.science/api/pith-number/YY4UVKZWWBWBVLXPEV52QQPXYC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC/action/storage_attestation","attest_author":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC/action/author_attestation","sign_citation":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC/action/citation_signature","submit_replication":"https://pith.science/pith/YY4UVKZWWBWBVLXPEV52QQPXYC/action/replication_record"}},"created_at":"2026-07-05T08:13:43.650008+00:00","updated_at":"2026-07-05T08:13:43.650008+00:00"}