{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RQP6A5FRCKMODUL4LYBMJU6SBS","short_pith_number":"pith:RQP6A5FR","schema_version":"1.0","canonical_sha256":"8c1fe074b11298e1d17c5e02c4d3d20cb037e6abb85c22e7724f12f73d215a39","source":{"kind":"arxiv","id":"2503.01328","version":2},"attestation_state":"computed","paper":{"title":"PipeOffload: Improving Scalability of Pipeline Parallelism with Memory Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DC"],"primary_cat":"cs.LG","authors_text":"Guangxing Huang, Jialin Li, Min Lin, Penghui Qi, Xinyi Wan","submitted_at":"2025-03-03T09:11:06Z","abstract_excerpt":"Pipeline parallelism (PP) is widely used for training large language models (LLMs), yet its scalability is often constrained by high activation memory consumption as the number of in-flight microbatches grows with the degree of PP. In this paper, we focus on addressing this challenge by leveraging the under-explored memory offload strategy in PP. With empirical study, we discover that in the majority of standard configurations, at least half, and potentially all, of the activations can be offloaded with negligible overhead. In the cases where full overload is not possible, we introduce a novel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01328","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-03T09:11:06Z","cross_cats_sorted":["cs.AI","cs.DC"],"title_canon_sha256":"73dd25d38168398af05ca72d58baeffdc756a7dff8c1d1bf6a0a88defd38fb59","abstract_canon_sha256":"1ee3b444ba462f6c20acc40a6f5a693b8406d87ec538bb6050406ebb9f3035de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:07.534758Z","signature_b64":"7T/A7BORqtzgL7OxQ8wirk9qf6UVSUOLXKVrgutv6bJ5uk+dFj/FbFV9GoJSmNEoArMTi/PbpmONmN612NI7DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c1fe074b11298e1d17c5e02c4d3d20cb037e6abb85c22e7724f12f73d215a39","last_reissued_at":"2026-07-05T11:29:07.534187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:07.534187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PipeOffload: Improving Scalability of Pipeline Parallelism with Memory Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.DC"],"primary_cat":"cs.LG","authors_text":"Guangxing Huang, Jialin Li, Min Lin, Penghui Qi, Xinyi Wan","submitted_at":"2025-03-03T09:11:06Z","abstract_excerpt":"Pipeline parallelism (PP) is widely used for training large language models (LLMs), yet its scalability is often constrained by high activation memory consumption as the number of in-flight microbatches grows with the degree of PP. In this paper, we focus on addressing this challenge by leveraging the under-explored memory offload strategy in PP. With empirical study, we discover that in the majority of standard configurations, at least half, and potentially all, of the activations can be offloaded with negligible overhead. In the cases where full overload is not possible, we introduce a novel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01328","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01328/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01328","created_at":"2026-07-05T11:29:07.534248+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01328v2","created_at":"2026-07-05T11:29:07.534248+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01328","created_at":"2026-07-05T11:29:07.534248+00:00"},{"alias_kind":"pith_short_12","alias_value":"RQP6A5FRCKMO","created_at":"2026-07-05T11:29:07.534248+00:00"},{"alias_kind":"pith_short_16","alias_value":"RQP6A5FRCKMODUL4","created_at":"2026-07-05T11:29:07.534248+00:00"},{"alias_kind":"pith_short_8","alias_value":"RQP6A5FR","created_at":"2026-07-05T11:29:07.534248+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00539","citing_title":"GNMR: Runtime Stability Control for Low-Precision Large Language Model Training","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02189","citing_title":"PipeMax: Enhancing Offline LLM Inference on Commodity GPU Servers","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS","json":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS.json","graph_json":"https://pith.science/api/pith-number/RQP6A5FRCKMODUL4LYBMJU6SBS/graph.json","events_json":"https://pith.science/api/pith-number/RQP6A5FRCKMODUL4LYBMJU6SBS/events.json","paper":"https://pith.science/paper/RQP6A5FR"},"agent_actions":{"view_html":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS","download_json":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS.json","view_paper":"https://pith.science/paper/RQP6A5FR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01328&json=true","fetch_graph":"https://pith.science/api/pith-number/RQP6A5FRCKMODUL4LYBMJU6SBS/graph.json","fetch_events":"https://pith.science/api/pith-number/RQP6A5FRCKMODUL4LYBMJU6SBS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS/action/storage_attestation","attest_author":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS/action/author_attestation","sign_citation":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS/action/citation_signature","submit_replication":"https://pith.science/pith/RQP6A5FRCKMODUL4LYBMJU6SBS/action/replication_record"}},"created_at":"2026-07-05T11:29:07.534248+00:00","updated_at":"2026-07-05T11:29:07.534248+00:00"}