{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UMD2PP2V3A3OA7E5UWGEQPR667","short_pith_number":"pith:UMD2PP2V","schema_version":"1.0","canonical_sha256":"a307a7bf55d836e07c9da58c483e3ef7cd1a2dde42f8b028b758b8c380118a9f","source":{"kind":"arxiv","id":"2404.05019","version":3},"attestation_state":"computed","paper":{"title":"Shortcut-connected Expert Parallelism for Accelerating Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Jiayi Huang, Junwei Cui, Juyong Jiang, Le Qin, Sunghun Kim, Weilin Cai","submitted_at":"2024-04-07T17:17:23Z","abstract_excerpt":"Expert parallelism has emerged as a key strategy for distributing the computational workload of sparsely-gated mixture-of-experts (MoE) models across multiple devices, enabling the processing of increasingly large-scale models. However, the All-to-All communication inherent to expert parallelism poses a significant bottleneck, limiting the efficiency of MoE models. Although existing optimization methods partially mitigate this issue, they remain constrained by the sequential dependency between communication and computation operations. To address this challenge, we propose ScMoE, a novel shortc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05019","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-07T17:17:23Z","cross_cats_sorted":["cs.CL","cs.DC"],"title_canon_sha256":"aa837f9b7bedfd98e585cab788e899ae17ec348303f40a0d5a6c2abc4b70c659","abstract_canon_sha256":"7d329114fe55da3d43abd46e647e3a2cab9fdc951a70c353103e9292ce479627"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:31.690449Z","signature_b64":"WETNY+jIsw68+Pp5GnGWLzpeShb+Mbi28I6S2A7x13KQpWvkI6iyOtU0jDGNA9SrvIexoAn7gKobAf+YyN8uDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a307a7bf55d836e07c9da58c483e3ef7cd1a2dde42f8b028b758b8c380118a9f","last_reissued_at":"2026-07-05T11:11:31.689882Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:31.689882Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Shortcut-connected Expert Parallelism for Accelerating Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Jiayi Huang, Junwei Cui, Juyong Jiang, Le Qin, Sunghun Kim, Weilin Cai","submitted_at":"2024-04-07T17:17:23Z","abstract_excerpt":"Expert parallelism has emerged as a key strategy for distributing the computational workload of sparsely-gated mixture-of-experts (MoE) models across multiple devices, enabling the processing of increasingly large-scale models. However, the All-to-All communication inherent to expert parallelism poses a significant bottleneck, limiting the efficiency of MoE models. Although existing optimization methods partially mitigate this issue, they remain constrained by the sequential dependency between communication and computation operations. To address this challenge, we propose ScMoE, a novel shortc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05019","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05019/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05019","created_at":"2026-07-05T11:11:31.689945+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05019v3","created_at":"2026-07-05T11:11:31.689945+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05019","created_at":"2026-07-05T11:11:31.689945+00:00"},{"alias_kind":"pith_short_12","alias_value":"UMD2PP2V3A3O","created_at":"2026-07-05T11:11:31.689945+00:00"},{"alias_kind":"pith_short_16","alias_value":"UMD2PP2V3A3OA7E5","created_at":"2026-07-05T11:11:31.689945+00:00"},{"alias_kind":"pith_short_8","alias_value":"UMD2PP2V","created_at":"2026-07-05T11:11:31.689945+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26842","citing_title":"MONA: Muon Optimizer with Nesterov Acceleration for Scalable Language Model Training","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21613","citing_title":"Chameleon: Adaptive Fault Tolerance for Distributed Training via Real-time Policy Selection","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667","json":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667.json","graph_json":"https://pith.science/api/pith-number/UMD2PP2V3A3OA7E5UWGEQPR667/graph.json","events_json":"https://pith.science/api/pith-number/UMD2PP2V3A3OA7E5UWGEQPR667/events.json","paper":"https://pith.science/paper/UMD2PP2V"},"agent_actions":{"view_html":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667","download_json":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667.json","view_paper":"https://pith.science/paper/UMD2PP2V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05019&json=true","fetch_graph":"https://pith.science/api/pith-number/UMD2PP2V3A3OA7E5UWGEQPR667/graph.json","fetch_events":"https://pith.science/api/pith-number/UMD2PP2V3A3OA7E5UWGEQPR667/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667/action/storage_attestation","attest_author":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667/action/author_attestation","sign_citation":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667/action/citation_signature","submit_replication":"https://pith.science/pith/UMD2PP2V3A3OA7E5UWGEQPR667/action/replication_record"}},"created_at":"2026-07-05T11:11:31.689945+00:00","updated_at":"2026-07-05T11:11:31.689945+00:00"}