{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HCIUEKOCZK7GBZAYZMPYYDJVV7","short_pith_number":"pith:HCIUEKOC","schema_version":"1.0","canonical_sha256":"38914229c2cabe60e418cb1f8c0d35afdf4bb1219fc2c0466ceba69676d4e042","source":{"kind":"arxiv","id":"2103.13262","version":1},"attestation_state":"computed","paper":{"title":"FastMoE: A Fast Mixture-of-Expert Training System","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Aohan Zeng, Jiaao He, Jidong Zhai, Jie Tang, Jiezhong Qiu, Zhilin Yang","submitted_at":"2021-03-24T15:27:15Z","abstract_excerpt":"Mixture-of-Expert (MoE) presents a strong potential in enlarging the size of language model to trillions of parameters. However, training trillion-scale MoE requires algorithm and system co-design for a well-tuned high performance distributed training system. Unfortunately, the only existing platform that meets the requirements strongly depends on Google's hardware (TPU) and software (Mesh Tensorflow) stack, and is not open and available to the public, especially GPU and PyTorch communities.\n  In this paper, we present FastMoE, a distributed MoE training system based on PyTorch with common acc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.13262","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-24T15:27:15Z","cross_cats_sorted":["cs.CL","cs.DC"],"title_canon_sha256":"420eb6c48720380f673ee115386e26397350d2fc81cf3a26863e8764201dc24e","abstract_canon_sha256":"02d7aa237a82911dc057f44969c8d7d1f6ece3acbefa02c63d94102193813883"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:26:07.671155Z","signature_b64":"JO/pxPln58vzvoAz57Prr6mFrMFgfOs8fp3EwT42UeXMkjFF0m2Zj3U7CmUXlA/ADf7+RmbnS16AcdqkXEJ4Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38914229c2cabe60e418cb1f8c0d35afdf4bb1219fc2c0466ceba69676d4e042","last_reissued_at":"2026-07-05T02:26:07.670617Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:26:07.670617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FastMoE: A Fast Mixture-of-Expert Training System","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Aohan Zeng, Jiaao He, Jidong Zhai, Jie Tang, Jiezhong Qiu, Zhilin Yang","submitted_at":"2021-03-24T15:27:15Z","abstract_excerpt":"Mixture-of-Expert (MoE) presents a strong potential in enlarging the size of language model to trillions of parameters. However, training trillion-scale MoE requires algorithm and system co-design for a well-tuned high performance distributed training system. Unfortunately, the only existing platform that meets the requirements strongly depends on Google's hardware (TPU) and software (Mesh Tensorflow) stack, and is not open and available to the public, especially GPU and PyTorch communities.\n  In this paper, we present FastMoE, a distributed MoE training system based on PyTorch with common acc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.13262","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.13262/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.13262","created_at":"2026-07-05T02:26:07.670675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.13262v1","created_at":"2026-07-05T02:26:07.670675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.13262","created_at":"2026-07-05T02:26:07.670675+00:00"},{"alias_kind":"pith_short_12","alias_value":"HCIUEKOCZK7G","created_at":"2026-07-05T02:26:07.670675+00:00"},{"alias_kind":"pith_short_16","alias_value":"HCIUEKOCZK7GBZAY","created_at":"2026-07-05T02:26:07.670675+00:00"},{"alias_kind":"pith_short_8","alias_value":"HCIUEKOC","created_at":"2026-07-05T02:26:07.670675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08250","citing_title":"On the Design of Mixture-of-Experts for Dynamic Gaussian Splatting","ref_index":63,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31363","citing_title":"Language-Assisted Super-Resolution from Real-World Low-Resolution Patches","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00573","citing_title":"BrainFIBRE: A Foundation Model via Information Decomposition for Brain Microstructure","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00735","citing_title":"ViBE: Co-Optimizing Workload Skew and Hardware Variability for MoE Serving","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31363","citing_title":"Language-Assisted Super-Resolution from Real-World Low-Resolution Patches","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":218,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02960","citing_title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13247","citing_title":"EMO: Frustratingly Easy Progressive Training of Extendable MoE","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13247","citing_title":"EMO: Frustratingly Easy Progressive Training of Extendable MoE","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11537","citing_title":"Fast MoE Inference via Predictive Prefetching and Expert Replication","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2309.14509","citing_title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","ref_index":172,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00539","citing_title":"AGoQ: Activation and Gradient Quantization for Memory-Efficient Distributed Training of LLMs","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08292","citing_title":"Hierarchical Mixture-of-Experts with Two-Stage Optimization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00539","citing_title":"AGoQ: Activation and Gradient Quantization for Memory-Efficient Distributed Training of LLMs","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":211,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02960","citing_title":"MoE-Prefill: Zero Redundancy Overheads in MoE Prefill Serving","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7","json":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7.json","graph_json":"https://pith.science/api/pith-number/HCIUEKOCZK7GBZAYZMPYYDJVV7/graph.json","events_json":"https://pith.science/api/pith-number/HCIUEKOCZK7GBZAYZMPYYDJVV7/events.json","paper":"https://pith.science/paper/HCIUEKOC"},"agent_actions":{"view_html":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7","download_json":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7.json","view_paper":"https://pith.science/paper/HCIUEKOC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.13262&json=true","fetch_graph":"https://pith.science/api/pith-number/HCIUEKOCZK7GBZAYZMPYYDJVV7/graph.json","fetch_events":"https://pith.science/api/pith-number/HCIUEKOCZK7GBZAYZMPYYDJVV7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7/action/storage_attestation","attest_author":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7/action/author_attestation","sign_citation":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7/action/citation_signature","submit_replication":"https://pith.science/pith/HCIUEKOCZK7GBZAYZMPYYDJVV7/action/replication_record"}},"created_at":"2026-07-05T02:26:07.670675+00:00","updated_at":"2026-07-05T02:26:07.670675+00:00"}