{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UHFQFZGHTX4BF3BZWUV6CGYNTC","short_pith_number":"pith:UHFQFZGH","schema_version":"1.0","canonical_sha256":"a1cb02e4c79df812ec39b52be11b0d98ab46239bb2f5c8bbeb448acc2445f5f0","source":{"kind":"arxiv","id":"2505.05713","version":2},"attestation_state":"computed","paper":{"title":"Understanding Stragglers in Large Model Training Using What-if Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Aurojit Panda, Chenyuan Wang, Haibin Lin, Jinkun Lin, Jinyang Li, Menghan Yu, Shuguang Wang, Sida Zhao, Wei Jia, Xiang Shi, Xin Liu, Zhanghan Wang, Zherui Liu, Ziheng Jiang, Zuocheng Shi, Zuquan Song","submitted_at":"2025-05-09T01:24:24Z","abstract_excerpt":"Large language model (LLM) training is one of the most demanding distributed computations today, often requiring thousands of GPUs with frequent synchronization across machines. Such a workload pattern makes it susceptible to stragglers, where the training can be stalled by few slow workers. At ByteDance we find stragglers are not trivially always caused by hardware failures, but can arise from multiple complex factors. This work aims to present a comprehensive study on the straggler issues in LLM training, using a five-month trace collected from our ByteDance LLM training cluster. The core me"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.05713","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2025-05-09T01:24:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9d572983e2d637d5a9929d63644f755a110757252ee828884835ca5d323bb464","abstract_canon_sha256":"ddf2e93b0b0f7b936b0154d53370e10cc3818a7d4155b192d0c1cbca864eed3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:37.087447Z","signature_b64":"HaqbZg9MVcFr10bTC1i4/Q1zvSn1G81FALwiO/ESq6ZcvloAbvjMl6olBduBjOXWKYSgz4uc4YZVz+QBv0iFDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1cb02e4c79df812ec39b52be11b0d98ab46239bb2f5c8bbeb448acc2445f5f0","last_reissued_at":"2026-07-05T11:01:37.086934Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:37.086934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Stragglers in Large Model Training Using What-if Analysis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Aurojit Panda, Chenyuan Wang, Haibin Lin, Jinkun Lin, Jinyang Li, Menghan Yu, Shuguang Wang, Sida Zhao, Wei Jia, Xiang Shi, Xin Liu, Zhanghan Wang, Zherui Liu, Ziheng Jiang, Zuocheng Shi, Zuquan Song","submitted_at":"2025-05-09T01:24:24Z","abstract_excerpt":"Large language model (LLM) training is one of the most demanding distributed computations today, often requiring thousands of GPUs with frequent synchronization across machines. Such a workload pattern makes it susceptible to stragglers, where the training can be stalled by few slow workers. At ByteDance we find stragglers are not trivially always caused by hardware failures, but can arise from multiple complex factors. This work aims to present a comprehensive study on the straggler issues in LLM training, using a five-month trace collected from our ByteDance LLM training cluster. The core me"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.05713","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.05713/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.05713","created_at":"2026-07-05T11:01:37.086994+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.05713v2","created_at":"2026-07-05T11:01:37.086994+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.05713","created_at":"2026-07-05T11:01:37.086994+00:00"},{"alias_kind":"pith_short_12","alias_value":"UHFQFZGHTX4B","created_at":"2026-07-05T11:01:37.086994+00:00"},{"alias_kind":"pith_short_16","alias_value":"UHFQFZGHTX4BF3BZ","created_at":"2026-07-05T11:01:37.086994+00:00"},{"alias_kind":"pith_short_8","alias_value":"UHFQFZGH","created_at":"2026-07-05T11:01:37.086994+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09370","citing_title":"From Detection to Recovery: Operational Analysis on LLM Pre-training with 504 GPUs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00831","citing_title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06374","citing_title":"ResiHP: Taming LLM Training Failures with Dynamic Hybrid Parallelism","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09370","citing_title":"From Detection to Recovery: Operational Analysis on LLM Pre-training with 504 GPUs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06374","citing_title":"ResiHP: Taming LLM Training Failures with Dynamic Hybrid Parallelism","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC","json":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC.json","graph_json":"https://pith.science/api/pith-number/UHFQFZGHTX4BF3BZWUV6CGYNTC/graph.json","events_json":"https://pith.science/api/pith-number/UHFQFZGHTX4BF3BZWUV6CGYNTC/events.json","paper":"https://pith.science/paper/UHFQFZGH"},"agent_actions":{"view_html":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC","download_json":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC.json","view_paper":"https://pith.science/paper/UHFQFZGH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.05713&json=true","fetch_graph":"https://pith.science/api/pith-number/UHFQFZGHTX4BF3BZWUV6CGYNTC/graph.json","fetch_events":"https://pith.science/api/pith-number/UHFQFZGHTX4BF3BZWUV6CGYNTC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC/action/storage_attestation","attest_author":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC/action/author_attestation","sign_citation":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC/action/citation_signature","submit_replication":"https://pith.science/pith/UHFQFZGHTX4BF3BZWUV6CGYNTC/action/replication_record"}},"created_at":"2026-07-05T11:01:37.086994+00:00","updated_at":"2026-07-05T11:01:37.086994+00:00"}