{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FJ5FKV6QEVEXNXE6L2YA6VKBUT","short_pith_number":"pith:FJ5FKV6Q","schema_version":"1.0","canonical_sha256":"2a7a5557d0254976dc9e5eb00f5541a4dea7e0c0953cf6bb8171fddda01a89a1","source":{"kind":"arxiv","id":"2411.01288","version":4},"attestation_state":"computed","paper":{"title":"Hexa-MoE: Efficient and Heterogeneous-aware Training for Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Hanrui Wang, Jie Peng, Pingzhi Li, Shuqing Luo, Tianlong Chen","submitted_at":"2024-11-02T15:45:54Z","abstract_excerpt":"Mixture-of-Experts (MoE) has emerged as a practical approach to scale up parameters for the Transformer model to achieve better generalization while maintaining a sub-linear increase in computation overhead. Current MoE models are mainly built with expert parallelism on distributed devices. However, it usually depends on homogeneous devices to deploy and suffers from heavy communication overhead and computation redundancy. In this paper, we explore developing a \\texttt{H}eterogeneous-aware \\texttt{EX}pert \\texttt{A}llocation framework, \\textbf{\\texttt{HEXA-MoE}}, with significantly enhanced co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01288","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-11-02T15:45:54Z","cross_cats_sorted":[],"title_canon_sha256":"c137273b872737a763758cd5ae43860e1e681b22112e88a05bfec0b4816b4529","abstract_canon_sha256":"8292f013186727dfd71fee49570fd2bf21e6851a6f4fac1ec82c275d8509bfff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:02.929588Z","signature_b64":"GsC4D2LO1I8amxQMglwe9XFE5FeWpQW1O+PR2evivVILP9DWDy9TeZ4pMGydwiPGPMxlWS4w7TYhAMR+TaYPDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a7a5557d0254976dc9e5eb00f5541a4dea7e0c0953cf6bb8171fddda01a89a1","last_reissued_at":"2026-07-05T10:43:02.929081Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:02.929081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hexa-MoE: Efficient and Heterogeneous-aware Training for Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Hanrui Wang, Jie Peng, Pingzhi Li, Shuqing Luo, Tianlong Chen","submitted_at":"2024-11-02T15:45:54Z","abstract_excerpt":"Mixture-of-Experts (MoE) has emerged as a practical approach to scale up parameters for the Transformer model to achieve better generalization while maintaining a sub-linear increase in computation overhead. Current MoE models are mainly built with expert parallelism on distributed devices. However, it usually depends on homogeneous devices to deploy and suffers from heavy communication overhead and computation redundancy. In this paper, we explore developing a \\texttt{H}eterogeneous-aware \\texttt{EX}pert \\texttt{A}llocation framework, \\textbf{\\texttt{HEXA-MoE}}, with significantly enhanced co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01288","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01288/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01288","created_at":"2026-07-05T10:43:02.929140+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01288v4","created_at":"2026-07-05T10:43:02.929140+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01288","created_at":"2026-07-05T10:43:02.929140+00:00"},{"alias_kind":"pith_short_12","alias_value":"FJ5FKV6QEVEX","created_at":"2026-07-05T10:43:02.929140+00:00"},{"alias_kind":"pith_short_16","alias_value":"FJ5FKV6QEVEXNXE6","created_at":"2026-07-05T10:43:02.929140+00:00"},{"alias_kind":"pith_short_8","alias_value":"FJ5FKV6Q","created_at":"2026-07-05T10:43:02.929140+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT","json":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT.json","graph_json":"https://pith.science/api/pith-number/FJ5FKV6QEVEXNXE6L2YA6VKBUT/graph.json","events_json":"https://pith.science/api/pith-number/FJ5FKV6QEVEXNXE6L2YA6VKBUT/events.json","paper":"https://pith.science/paper/FJ5FKV6Q"},"agent_actions":{"view_html":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT","download_json":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT.json","view_paper":"https://pith.science/paper/FJ5FKV6Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01288&json=true","fetch_graph":"https://pith.science/api/pith-number/FJ5FKV6QEVEXNXE6L2YA6VKBUT/graph.json","fetch_events":"https://pith.science/api/pith-number/FJ5FKV6QEVEXNXE6L2YA6VKBUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT/action/storage_attestation","attest_author":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT/action/author_attestation","sign_citation":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT/action/citation_signature","submit_replication":"https://pith.science/pith/FJ5FKV6QEVEXNXE6L2YA6VKBUT/action/replication_record"}},"created_at":"2026-07-05T10:43:02.929140+00:00","updated_at":"2026-07-05T10:43:02.929140+00:00"}