{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XR2UHLWNHHFSD4ZJJXWRU6R6QP","short_pith_number":"pith:XR2UHLWN","schema_version":"1.0","canonical_sha256":"bc7543aecd39cb21f3294ded1a7a3e83e3d13c5d82445f0e623f0f8f8faf88f6","source":{"kind":"arxiv","id":"2402.15627","version":1},"attestation_state":"computed","paper":{"title":"MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Cong Xie, Ding Zhou, Haibin Lin, Haohan Xu, Haoran Wei, Hongmin Chen, Jianxi Ye, Leqi Zou, Liang Xiang, Pengfei Nie, Qi Hou, Qi Huang, Shibiao Nong, Shipeng Yan, Sida Zhao, Sun He, Xiang Li, Xiaoying Jia, Xin Jin, Xin Liu, Yanghua Peng, Yangrui Chen, Yinmin Zhong, Yiyao Sheng, Yulu Jia, Zhang Zhang, Zhe Li, Zherui Liu, Zhihao Bai, Zhi Zhang, Zhuo Jiang, Ziheng Jiang","submitted_at":"2024-02-23T22:10:59Z","abstract_excerpt":"We present the design, implementation and engineering experience in building and deploying MegaScale, a production system for training large language models (LLMs) at the scale of more than 10,000 GPUs. Training LLMs at this scale brings unprecedented challenges to training efficiency and stability. We take a full-stack approach that co-designs the algorithmic and system components across model block and optimizer design, computation and communication overlapping, operator optimization, data pipeline, and network performance tuning. Maintaining high efficiency throughout the training process ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.15627","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-23T22:10:59Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"965ecaee53eed9d5aab76f4c165b841a1f89677bff99362b775bbc76a9d427f7","abstract_canon_sha256":"ce0c3f4ade76bd25b3ab5b2cd3576980c8a1f3e0d142a7c6f3f912088a62ca84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:55.792980Z","signature_b64":"F0IC0Up+BemLS818jLwPZmxApldJ0bFp0RopVT8Nz0rNNXInXdfmbz5LfJ81ZuhaKeKK9zRN/vNnfIz4Zs/eCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc7543aecd39cb21f3294ded1a7a3e83e3d13c5d82445f0e623f0f8f8faf88f6","last_reissued_at":"2026-07-05T07:48:55.792483Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:55.792483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Cong Xie, Ding Zhou, Haibin Lin, Haohan Xu, Haoran Wei, Hongmin Chen, Jianxi Ye, Leqi Zou, Liang Xiang, Pengfei Nie, Qi Hou, Qi Huang, Shibiao Nong, Shipeng Yan, Sida Zhao, Sun He, Xiang Li, Xiaoying Jia, Xin Jin, Xin Liu, Yanghua Peng, Yangrui Chen, Yinmin Zhong, Yiyao Sheng, Yulu Jia, Zhang Zhang, Zhe Li, Zherui Liu, Zhihao Bai, Zhi Zhang, Zhuo Jiang, Ziheng Jiang","submitted_at":"2024-02-23T22:10:59Z","abstract_excerpt":"We present the design, implementation and engineering experience in building and deploying MegaScale, a production system for training large language models (LLMs) at the scale of more than 10,000 GPUs. Training LLMs at this scale brings unprecedented challenges to training efficiency and stability. We take a full-stack approach that co-designs the algorithmic and system components across model block and optimizer design, computation and communication overlapping, operator optimization, data pipeline, and network performance tuning. Maintaining high efficiency throughout the training process ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.15627","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.15627/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.15627","created_at":"2026-07-05T07:48:55.792543+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.15627v1","created_at":"2026-07-05T07:48:55.792543+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.15627","created_at":"2026-07-05T07:48:55.792543+00:00"},{"alias_kind":"pith_short_12","alias_value":"XR2UHLWNHHFS","created_at":"2026-07-05T07:48:55.792543+00:00"},{"alias_kind":"pith_short_16","alias_value":"XR2UHLWNHHFSD4ZJ","created_at":"2026-07-05T07:48:55.792543+00:00"},{"alias_kind":"pith_short_8","alias_value":"XR2UHLWN","created_at":"2026-07-05T07:48:55.792543+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.14430","citing_title":"PipeFusion: Patch-level Pipeline Parallelism for Diffusion Transformers Inference","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20799","citing_title":"Instant GPU Efficiency Visibility at Fleet Scale","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2508.21613","citing_title":"Chameleon: Adaptive Fault Tolerance for Distributed Training via Real-time Policy Selection","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08962","citing_title":"MegaScale-Omni: A Hyper-Scale, Workload-Resilient System for MultiModal LLM Training in Production","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24088","citing_title":"TACO: Efficient Communication Compression of Intermediate Tensors for Scalable Tensor-Parallel LLM Training","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2409.19256","citing_title":"HybridFlow: A Flexible and Efficient RLHF Framework","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP","json":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP.json","graph_json":"https://pith.science/api/pith-number/XR2UHLWNHHFSD4ZJJXWRU6R6QP/graph.json","events_json":"https://pith.science/api/pith-number/XR2UHLWNHHFSD4ZJJXWRU6R6QP/events.json","paper":"https://pith.science/paper/XR2UHLWN"},"agent_actions":{"view_html":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP","download_json":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP.json","view_paper":"https://pith.science/paper/XR2UHLWN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.15627&json=true","fetch_graph":"https://pith.science/api/pith-number/XR2UHLWNHHFSD4ZJJXWRU6R6QP/graph.json","fetch_events":"https://pith.science/api/pith-number/XR2UHLWNHHFSD4ZJJXWRU6R6QP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP/action/storage_attestation","attest_author":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP/action/author_attestation","sign_citation":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP/action/citation_signature","submit_replication":"https://pith.science/pith/XR2UHLWNHHFSD4ZJJXWRU6R6QP/action/replication_record"}},"created_at":"2026-07-05T07:48:55.792543+00:00","updated_at":"2026-07-05T07:48:55.792543+00:00"}