{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JTCQ5Y7SHZQ6BCQKQLJFW2ERUN","short_pith_number":"pith:JTCQ5Y7S","schema_version":"1.0","canonical_sha256":"4cc50ee3f23e61e08a0a82d25b6891a374837dfa8140ff421d5320e8850c26bd","source":{"kind":"arxiv","id":"2406.04594","version":2},"attestation_state":"computed","paper":{"title":"Enhancing Large-Scale AI Training Efficiency: The C4 Solution for Real-Time Anomaly Detection and Communication Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Ang Liu, Bin Luo, Binzhang Fu, Chang Zhou, Dennis Cai, Ennan Zhai, Fei Feng, Gang Lu, Hairong Jiao, Hanyu Zhao, Jiamang Wang, Jianbo Dong, Jianwei Zhang, Jun Zhang, Man Yuan, Pengcheng Zhang, Rui Men, Siran Yang, Wencong Xiao, Xiang Li, Yikai Zhu, Yi Shi, Yuan Xie, Yu Guan, Zian Chen","submitted_at":"2024-06-07T02:58:35Z","abstract_excerpt":"The emergence of Large Language Models (LLMs) has necessitated the adoption of distributed training techniques, involving the deployment of thousands of GPUs to train a single model. Unfortunately, the efficiency of large-scale distributed training systems is often suboptimal due to the increased likelihood of hardware errors in high-end GPU products and the heightened risk of network traffic collisions. Moreover, any local hardware failure can disrupt training tasks, and the inability to swiftly identify faulty components leads to a significant waste of GPU resources. And, prolonged communica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04594","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-06-07T02:58:35Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"126f53911356ffcb67a90da8583407069893e8d2db3bce0b7094be63c2af136f","abstract_canon_sha256":"e803001d47da6a6df71b52bb913f38b7b182ea8c447abbba045cb37ef1b629ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:57.981084Z","signature_b64":"zM710q+GNwLBh7b+xh3NnY2ow7Oo1k/Fy0pzsilOArCGAwXNdNcDvqbNWTsRF/tylCY1HZORD4/rhkQP4KWqDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4cc50ee3f23e61e08a0a82d25b6891a374837dfa8140ff421d5320e8850c26bd","last_reissued_at":"2026-07-05T11:07:57.980576Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:57.980576Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Large-Scale AI Training Efficiency: The C4 Solution for Real-Time Anomaly Detection and Communication Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Ang Liu, Bin Luo, Binzhang Fu, Chang Zhou, Dennis Cai, Ennan Zhai, Fei Feng, Gang Lu, Hairong Jiao, Hanyu Zhao, Jiamang Wang, Jianbo Dong, Jianwei Zhang, Jun Zhang, Man Yuan, Pengcheng Zhang, Rui Men, Siran Yang, Wencong Xiao, Xiang Li, Yikai Zhu, Yi Shi, Yuan Xie, Yu Guan, Zian Chen","submitted_at":"2024-06-07T02:58:35Z","abstract_excerpt":"The emergence of Large Language Models (LLMs) has necessitated the adoption of distributed training techniques, involving the deployment of thousands of GPUs to train a single model. Unfortunately, the efficiency of large-scale distributed training systems is often suboptimal due to the increased likelihood of hardware errors in high-end GPU products and the heightened risk of network traffic collisions. Moreover, any local hardware failure can disrupt training tasks, and the inability to swiftly identify faulty components leads to a significant waste of GPU resources. And, prolonged communica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04594","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04594/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04594","created_at":"2026-07-05T11:07:57.980637+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04594v2","created_at":"2026-07-05T11:07:57.980637+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04594","created_at":"2026-07-05T11:07:57.980637+00:00"},{"alias_kind":"pith_short_12","alias_value":"JTCQ5Y7SHZQ6","created_at":"2026-07-05T11:07:57.980637+00:00"},{"alias_kind":"pith_short_16","alias_value":"JTCQ5Y7SHZQ6BCQK","created_at":"2026-07-05T11:07:57.980637+00:00"},{"alias_kind":"pith_short_8","alias_value":"JTCQ5Y7S","created_at":"2026-07-05T11:07:57.980637+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18683","citing_title":"EPIC: Abstraction and Polymorphism of In-Network Collectives on Ethernet","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.16056","citing_title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN","json":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN.json","graph_json":"https://pith.science/api/pith-number/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/graph.json","events_json":"https://pith.science/api/pith-number/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/events.json","paper":"https://pith.science/paper/JTCQ5Y7S"},"agent_actions":{"view_html":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN","download_json":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN.json","view_paper":"https://pith.science/paper/JTCQ5Y7S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04594&json=true","fetch_graph":"https://pith.science/api/pith-number/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/graph.json","fetch_events":"https://pith.science/api/pith-number/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/action/storage_attestation","attest_author":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/action/author_attestation","sign_citation":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/action/citation_signature","submit_replication":"https://pith.science/pith/JTCQ5Y7SHZQ6BCQKQLJFW2ERUN/action/replication_record"}},"created_at":"2026-07-05T11:07:57.980637+00:00","updated_at":"2026-07-05T11:07:57.980637+00:00"}