{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:MABUVFA27T25GAGIVF43WJIDVG","short_pith_number":"pith:MABUVFA2","schema_version":"1.0","canonical_sha256":"60034a941afcf5d300c8a979bb2503a994b7bc7fd3bad1fb89cdc40341290e0c","source":{"kind":"arxiv","id":"2203.14685","version":3},"attestation_state":"computed","paper":{"title":"HetuMoE: An Efficient Trillion-scale Mixture-of-Expert Distributed Training System","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Bin Cui, Pinxue Zhao, Tong Zhao, Xiaonan Nie, Xupeng Miao","submitted_at":"2022-03-28T12:32:25Z","abstract_excerpt":"As giant dense models advance quality but require large amounts of GPU budgets for training, the sparsely gated Mixture-of-Experts (MoE), a kind of conditional computation architecture, is proposed to scale models while keeping their computation constant. Specifically, the input tokens are routed by the gate network and only activates part of the expert network. Existing MoE training systems only support part of mainstream MoE models (e.g. Top k) training under expensive high-bandwidth GPU clusters. In this paper, we present HetuMoE, a high-performance large-scale sparse MoE training system bu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.14685","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2022-03-28T12:32:25Z","cross_cats_sorted":[],"title_canon_sha256":"e91f74307a75e4aa45f826e022631c2b3b55a1bad1d82563b74c4d721cc7f6b1","abstract_canon_sha256":"308e09d978c0c18d660fbe666d13bf4276f7baf42775834b0d2c6054f87b5479"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:16:50.489459Z","signature_b64":"i7sFhs7cwGDpQFB8/d9jZtvbedRgFCtXuHhsFmHz1P4xjwPGVympKEi3rUvip9Bsqf8EJlj4+Kr/3djQUSc4BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60034a941afcf5d300c8a979bb2503a994b7bc7fd3bad1fb89cdc40341290e0c","last_reissued_at":"2026-07-05T05:16:50.489055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:16:50.489055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HetuMoE: An Efficient Trillion-scale Mixture-of-Expert Distributed Training System","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Bin Cui, Pinxue Zhao, Tong Zhao, Xiaonan Nie, Xupeng Miao","submitted_at":"2022-03-28T12:32:25Z","abstract_excerpt":"As giant dense models advance quality but require large amounts of GPU budgets for training, the sparsely gated Mixture-of-Experts (MoE), a kind of conditional computation architecture, is proposed to scale models while keeping their computation constant. Specifically, the input tokens are routed by the gate network and only activates part of the expert network. Existing MoE training systems only support part of mainstream MoE models (e.g. Top k) training under expensive high-bandwidth GPU clusters. In this paper, we present HetuMoE, a high-performance large-scale sparse MoE training system bu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.14685","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.14685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.14685","created_at":"2026-07-05T05:16:50.489111+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.14685v3","created_at":"2026-07-05T05:16:50.489111+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.14685","created_at":"2026-07-05T05:16:50.489111+00:00"},{"alias_kind":"pith_short_12","alias_value":"MABUVFA27T25","created_at":"2026-07-05T05:16:50.489111+00:00"},{"alias_kind":"pith_short_16","alias_value":"MABUVFA27T25GAGI","created_at":"2026-07-05T05:16:50.489111+00:00"},{"alias_kind":"pith_short_8","alias_value":"MABUVFA2","created_at":"2026-07-05T05:16:50.489111+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00539","citing_title":"AGoQ: Activation and Gradient Quantization for Memory-Efficient Distributed Training of LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08639","citing_title":"ReLibra: Routing-Replay-Guided Load Balancing for MoE Training in Reinforcement Learning","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05607","citing_title":"Accelerating MoE with Dynamic In-Switch Computing on Multi-GPUs","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00539","citing_title":"AGoQ: Activation and Gradient Quantization for Memory-Efficient Distributed Training of LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05049","citing_title":"Piper: Efficient Large-Scale MoE Training via Resource Modeling and Pipelined Hybrid Parallelism","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG","json":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG.json","graph_json":"https://pith.science/api/pith-number/MABUVFA27T25GAGIVF43WJIDVG/graph.json","events_json":"https://pith.science/api/pith-number/MABUVFA27T25GAGIVF43WJIDVG/events.json","paper":"https://pith.science/paper/MABUVFA2"},"agent_actions":{"view_html":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG","download_json":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG.json","view_paper":"https://pith.science/paper/MABUVFA2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.14685&json=true","fetch_graph":"https://pith.science/api/pith-number/MABUVFA27T25GAGIVF43WJIDVG/graph.json","fetch_events":"https://pith.science/api/pith-number/MABUVFA27T25GAGIVF43WJIDVG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG/action/storage_attestation","attest_author":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG/action/author_attestation","sign_citation":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG/action/citation_signature","submit_replication":"https://pith.science/pith/MABUVFA27T25GAGIVF43WJIDVG/action/replication_record"}},"created_at":"2026-07-05T05:16:50.489111+00:00","updated_at":"2026-07-05T05:16:50.489111+00:00"}