{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DWDDCERYNRLWLCCIJPYJYWFRNJ","short_pith_number":"pith:DWDDCERY","schema_version":"1.0","canonical_sha256":"1d863112386c576588484bf09c58b16a598053d4d4dce1a757a36ca1f64b56de","source":{"kind":"arxiv","id":"2504.03871","version":1},"attestation_state":"computed","paper":{"title":"HeterMoE: Efficient Training of Mixture-of-Experts Models on Heterogeneous GPUs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Ceyu Xu, Danyang Zhuo, Feng Qian, Ion Stoica, Matthew Lentz, Shuowei Jin, Xueshen Liu, Yongji Wu, Z. Morley Mao","submitted_at":"2025-04-04T18:55:52Z","abstract_excerpt":"The Mixture-of-Experts (MoE) architecture has become increasingly popular as a method to scale up large language models (LLMs). To save costs, heterogeneity-aware training solutions have been proposed to utilize GPU clusters made up of both newer and older-generation GPUs. However, existing solutions are agnostic to the performance characteristics of different MoE model components (i.e., attention and expert) and do not fully utilize each GPU's compute capability.\n  In this paper, we introduce HeterMoE, a system to efficiently train MoE models on heterogeneous GPUs. Our key insight is that new"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.03871","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-04-04T18:55:52Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e8e18e3810a2fd09d8ba2290539816f63782e77d1f59f51dbef4abdb4ac66a8b","abstract_canon_sha256":"e1c517fb921c983e20754f72765af064ecec621c1b609b9144cbfb6b16e63168"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:50.080229Z","signature_b64":"qMB6uEi9y1JfHze8sO16A074+HhDRIf/sk94vNbyZu8RDAbp/C47yQHUDywMW5S9/0m72KITkHez+N3zBT+DDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d863112386c576588484bf09c58b16a598053d4d4dce1a757a36ca1f64b56de","last_reissued_at":"2026-07-05T10:44:50.079737Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:50.079737Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HeterMoE: Efficient Training of Mixture-of-Experts Models on Heterogeneous GPUs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Ceyu Xu, Danyang Zhuo, Feng Qian, Ion Stoica, Matthew Lentz, Shuowei Jin, Xueshen Liu, Yongji Wu, Z. Morley Mao","submitted_at":"2025-04-04T18:55:52Z","abstract_excerpt":"The Mixture-of-Experts (MoE) architecture has become increasingly popular as a method to scale up large language models (LLMs). To save costs, heterogeneity-aware training solutions have been proposed to utilize GPU clusters made up of both newer and older-generation GPUs. However, existing solutions are agnostic to the performance characteristics of different MoE model components (i.e., attention and expert) and do not fully utilize each GPU's compute capability.\n  In this paper, we introduce HeterMoE, a system to efficiently train MoE models on heterogeneous GPUs. Our key insight is that new"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.03871","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.03871/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.03871","created_at":"2026-07-05T10:44:50.079796+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.03871v1","created_at":"2026-07-05T10:44:50.079796+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.03871","created_at":"2026-07-05T10:44:50.079796+00:00"},{"alias_kind":"pith_short_12","alias_value":"DWDDCERYNRLW","created_at":"2026-07-05T10:44:50.079796+00:00"},{"alias_kind":"pith_short_16","alias_value":"DWDDCERYNRLWLCCI","created_at":"2026-07-05T10:44:50.079796+00:00"},{"alias_kind":"pith_short_8","alias_value":"DWDDCERY","created_at":"2026-07-05T10:44:50.079796+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26633","citing_title":"Simulating Unified Tensor Resharding in heterogeneous AI systems","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20982","citing_title":"Diagnosing Overhead in Dispatch Operations: Cross-architecture Observatory","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12476","citing_title":"HetRL: Efficient Reinforcement Learning for LLMs in Heterogeneous Environments","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11005","citing_title":"DisagMoE: Computation-Communication overlapped MoE Training via Disaggregated AF-Pipe Parallelism","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19241","citing_title":"UniEP: Unified Expert-Parallel MoE MegaKernel for LLM Training","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ","json":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ.json","graph_json":"https://pith.science/api/pith-number/DWDDCERYNRLWLCCIJPYJYWFRNJ/graph.json","events_json":"https://pith.science/api/pith-number/DWDDCERYNRLWLCCIJPYJYWFRNJ/events.json","paper":"https://pith.science/paper/DWDDCERY"},"agent_actions":{"view_html":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ","download_json":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ.json","view_paper":"https://pith.science/paper/DWDDCERY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.03871&json=true","fetch_graph":"https://pith.science/api/pith-number/DWDDCERYNRLWLCCIJPYJYWFRNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/DWDDCERYNRLWLCCIJPYJYWFRNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ/action/storage_attestation","attest_author":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ/action/author_attestation","sign_citation":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ/action/citation_signature","submit_replication":"https://pith.science/pith/DWDDCERYNRLWLCCIJPYJYWFRNJ/action/replication_record"}},"created_at":"2026-07-05T10:44:50.079796+00:00","updated_at":"2026-07-05T10:44:50.079796+00:00"}