{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:L5Y4YBEWHMNOUANEI723JO2ZA3","short_pith_number":"pith:L5Y4YBEW","schema_version":"1.0","canonical_sha256":"5f71cc04963b1aea01a447f5b4bb5906e1ba59bff32759fb8d840b89c4fa3ef6","source":{"kind":"arxiv","id":"2401.10241","version":1},"attestation_state":"computed","paper":{"title":"Zero Bubble Pipeline Parallelism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Guangxing Huang, Min Lin, Penghui Qi, Xinyi Wan","submitted_at":"2023-11-30T10:40:34Z","abstract_excerpt":"Pipeline parallelism is one of the key components for large-scale distributed training, yet its efficiency suffers from pipeline bubbles which were deemed inevitable. In this work, we introduce a scheduling strategy that, to our knowledge, is the first to successfully achieve zero pipeline bubbles under synchronous training semantics. The key idea behind this improvement is to split the backward computation into two parts, one that computes gradient for the input and another that computes for the parameters. Based on this idea, we handcraft novel pipeline schedules that significantly outperfor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.10241","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2023-11-30T10:40:34Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"658503bb13d8e47e87919bac743b250e47807757188c52b299a2aabb225fa8d8","abstract_canon_sha256":"5d531672a54c6525acecab73f31d16fcd76821904b7ed5a0b9907be70823d5bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:35:08.751525Z","signature_b64":"r5EWL3GXyet6dgSVIkw5GS1u3e8QUMe6cie3bgX5PJXnNU87dD3yN01BtxUKTeYYLDvsLNF3Afd1tuc+T5IYBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f71cc04963b1aea01a447f5b4bb5906e1ba59bff32759fb8d840b89c4fa3ef6","last_reissued_at":"2026-07-05T07:35:08.750724Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:35:08.750724Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero Bubble Pipeline Parallelism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Guangxing Huang, Min Lin, Penghui Qi, Xinyi Wan","submitted_at":"2023-11-30T10:40:34Z","abstract_excerpt":"Pipeline parallelism is one of the key components for large-scale distributed training, yet its efficiency suffers from pipeline bubbles which were deemed inevitable. In this work, we introduce a scheduling strategy that, to our knowledge, is the first to successfully achieve zero pipeline bubbles under synchronous training semantics. The key idea behind this improvement is to split the backward computation into two parts, one that computes gradient for the input and another that computes for the parameters. Based on this idea, we handcraft novel pipeline schedules that significantly outperfor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.10241","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.10241/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.10241","created_at":"2026-07-05T07:35:08.750785+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.10241v1","created_at":"2026-07-05T07:35:08.750785+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.10241","created_at":"2026-07-05T07:35:08.750785+00:00"},{"alias_kind":"pith_short_12","alias_value":"L5Y4YBEWHMNO","created_at":"2026-07-05T07:35:08.750785+00:00"},{"alias_kind":"pith_short_16","alias_value":"L5Y4YBEWHMNOUANE","created_at":"2026-07-05T07:35:08.750785+00:00"},{"alias_kind":"pith_short_8","alias_value":"L5Y4YBEW","created_at":"2026-07-05T07:35:08.750785+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":226,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11169","citing_title":"Piper: A Programmable Distributed Training System","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07881","citing_title":"Breaking the Bubble: Asynchronous Pipeline Parallel Training with Bounded Weight Inconsistency","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30634","citing_title":"One-Step Gradient Delay is Not a Barrier for Large-Scale Asynchronous Pipeline Parallel LLM Pretraining","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25451","citing_title":"BigMac: Breaking the Pareto Frontier of Compute and Memory in Multimodal LLM Training","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15622","citing_title":"Position: Zeroth-Order Optimization in Deep Learning Is Underexplored, Not Underpowered","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18750","citing_title":"A Readiness-Driven Runtime for Pipeline-Parallel Training under Runtime Variability","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00432","citing_title":"Does Math Reasoning Improve General LLM Capabilities? Understanding Transferability of LLM Reasoning","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24859","citing_title":"HARP: Orchestrating Automated Parallel Training on Heterogeneous GPU Clusters","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15596","citing_title":"PRISM: Probabilistic Runtime Insights and Scalable Performance Modeling for Large-Scale Distributed Training","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12131","citing_title":"BOOST: BOttleneck-Optimized Scalable Training Framework for Low-Rank Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03044","citing_title":"JoyAI-LLM Flash: Advancing Mid-Scale LLMs with Token Efficiency","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27085","citing_title":"Efficient Training on Multiple Consumer GPUs with RoundPipe","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08962","citing_title":"MegaScale-Omni: A Hyper-Scale, Workload-Resilient System for MultiModal LLM Training in Production","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2602.15763","citing_title":"GLM-5: from Vibe Coding to Agentic Engineering","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2405.04434","citing_title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2507.20534","citing_title":"Kimi K2: Open Agentic Intelligence","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16880","citing_title":"Symphony: Taming Step Misalignments in the Network for Ring-based Collective Operations","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05049","citing_title":"Piper: Efficient Large-Scale MoE Training via Resource Modeling and Pipelined Hybrid Parallelism","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3","json":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3.json","graph_json":"https://pith.science/api/pith-number/L5Y4YBEWHMNOUANEI723JO2ZA3/graph.json","events_json":"https://pith.science/api/pith-number/L5Y4YBEWHMNOUANEI723JO2ZA3/events.json","paper":"https://pith.science/paper/L5Y4YBEW"},"agent_actions":{"view_html":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3","download_json":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3.json","view_paper":"https://pith.science/paper/L5Y4YBEW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.10241&json=true","fetch_graph":"https://pith.science/api/pith-number/L5Y4YBEWHMNOUANEI723JO2ZA3/graph.json","fetch_events":"https://pith.science/api/pith-number/L5Y4YBEWHMNOUANEI723JO2ZA3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3/action/storage_attestation","attest_author":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3/action/author_attestation","sign_citation":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3/action/citation_signature","submit_replication":"https://pith.science/pith/L5Y4YBEWHMNOUANEI723JO2ZA3/action/replication_record"}},"created_at":"2026-07-05T07:35:08.750785+00:00","updated_at":"2026-07-05T07:35:08.750785+00:00"}