{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XHNNPWFNQ4ZPTVKFXVZADSBBOK","short_pith_number":"pith:XHNNPWFN","schema_version":"1.0","canonical_sha256":"b9dad7d8ad8732f9d545bd7201c82172bb6a11af285fbf7fc9d821627eadc140","source":{"kind":"arxiv","id":"2407.20018","version":1},"attestation_state":"computed","paper":{"title":"Efficient Training of Large Language Models on Distributed Infrastructures: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Dahua Lin, Guoteng Wang, Hang Yan, Jiangfei Duan, Lijuan Jiang, Peng Sun, Qinghao Hu, Qizhen Weng, Shuo Zhang, Tianwei Zhang, Wenwen Qu, Xingcheng Zhang, Xin Jin, Xipeng Qiu, Yonggang Wen, Zerui Wang","submitted_at":"2024-07-29T13:53:27Z","abstract_excerpt":"Large Language Models (LLMs) like GPT and LLaMA are revolutionizing the AI industry with their sophisticated capabilities. Training these models requires vast GPU clusters and significant computing time, posing major challenges in terms of scalability, efficiency, and reliability. This survey explores recent advancements in training systems for LLMs, including innovations in training infrastructure with AI accelerators, networking, storage, and scheduling. Additionally, the survey covers parallelism strategies, as well as optimizations for computation, communication, and memory in distributed "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20018","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-07-29T13:53:27Z","cross_cats_sorted":[],"title_canon_sha256":"1718908fb3fc165b3297c34daeedee3082060dede49c40e6c1080bd0c5d5a37f","abstract_canon_sha256":"feb5356251581caeef4a4489732513ff6231809b1a6fae3725e02a06592dc532"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:47.850867Z","signature_b64":"IlhB47WzxFtuFQuWZowwK+VJARCjnOuApXCYzSyw5p9OgVj5fAQGX/CC1YYSVBmj9eRiZB0nZIBNX7bkrKJJAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9dad7d8ad8732f9d545bd7201c82172bb6a11af285fbf7fc9d821627eadc140","last_reissued_at":"2026-07-05T08:49:47.850396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:47.850396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Training of Large Language Models on Distributed Infrastructures: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Dahua Lin, Guoteng Wang, Hang Yan, Jiangfei Duan, Lijuan Jiang, Peng Sun, Qinghao Hu, Qizhen Weng, Shuo Zhang, Tianwei Zhang, Wenwen Qu, Xingcheng Zhang, Xin Jin, Xipeng Qiu, Yonggang Wen, Zerui Wang","submitted_at":"2024-07-29T13:53:27Z","abstract_excerpt":"Large Language Models (LLMs) like GPT and LLaMA are revolutionizing the AI industry with their sophisticated capabilities. Training these models requires vast GPU clusters and significant computing time, posing major challenges in terms of scalability, efficiency, and reliability. This survey explores recent advancements in training systems for LLMs, including innovations in training infrastructure with AI accelerators, networking, storage, and scheduling. Additionally, the survey covers parallelism strategies, as well as optimizations for computation, communication, and memory in distributed "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20018","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20018/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20018","created_at":"2026-07-05T08:49:47.850455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20018v1","created_at":"2026-07-05T08:49:47.850455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20018","created_at":"2026-07-05T08:49:47.850455+00:00"},{"alias_kind":"pith_short_12","alias_value":"XHNNPWFNQ4ZP","created_at":"2026-07-05T08:49:47.850455+00:00"},{"alias_kind":"pith_short_16","alias_value":"XHNNPWFNQ4ZPTVKF","created_at":"2026-07-05T08:49:47.850455+00:00"},{"alias_kind":"pith_short_8","alias_value":"XHNNPWFN","created_at":"2026-07-05T08:49:47.850455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24013","citing_title":"CommFuse: Hiding Tail Latency via Communication Decomposition and Fusion for Distributed LLM Training","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23467","citing_title":"Hybrid JIT-CUDA Graph Optimization for Low-Latency Large Language Model Inference","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18909","citing_title":"ChipLight: Cross-Layer Optimization of Chiplet Design with Optical Interconnects for LLM Training","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK","json":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK.json","graph_json":"https://pith.science/api/pith-number/XHNNPWFNQ4ZPTVKFXVZADSBBOK/graph.json","events_json":"https://pith.science/api/pith-number/XHNNPWFNQ4ZPTVKFXVZADSBBOK/events.json","paper":"https://pith.science/paper/XHNNPWFN"},"agent_actions":{"view_html":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK","download_json":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK.json","view_paper":"https://pith.science/paper/XHNNPWFN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20018&json=true","fetch_graph":"https://pith.science/api/pith-number/XHNNPWFNQ4ZPTVKFXVZADSBBOK/graph.json","fetch_events":"https://pith.science/api/pith-number/XHNNPWFNQ4ZPTVKFXVZADSBBOK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK/action/storage_attestation","attest_author":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK/action/author_attestation","sign_citation":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK/action/citation_signature","submit_replication":"https://pith.science/pith/XHNNPWFNQ4ZPTVKFXVZADSBBOK/action/replication_record"}},"created_at":"2026-07-05T08:49:47.850455+00:00","updated_at":"2026-07-05T08:49:47.850455+00:00"}