{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:UWZJ7SSC2VJTJTXADLBTLYIHDD","short_pith_number":"pith:UWZJ7SSC","schema_version":"1.0","canonical_sha256":"a5b29fca42d55334cee01ac335e10718e34fafd351013f0eb83364f13b8fee06","source":{"kind":"arxiv","id":"2102.03161","version":2},"attestation_state":"computed","paper":{"title":"PipeTransformer: Automated Elastic Pipelining for Distributed Training of Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chaoyang He, Mahdi Soltanolkotabi, Salman Avestimehr, Shen Li","submitted_at":"2021-02-05T13:39:31Z","abstract_excerpt":"The size of Transformer models is growing at an unprecedented pace. It has only taken less than one year to reach trillion-level parameters after the release of GPT-3 (175B). Training such models requires both substantial engineering efforts and enormous computing resources, which are luxuries most research teams cannot afford. In this paper, we propose PipeTransformer, which leverages automated and elastic pipelining and data parallelism for efficient distributed training of Transformer models. PipeTransformer automatically adjusts the pipelining and data parallelism by identifying and freezi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.03161","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-05T13:39:31Z","cross_cats_sorted":[],"title_canon_sha256":"79a5f220f9181442bd9c6aa0fee5645a616165ce75fbcd5d5631e982d9686d91","abstract_canon_sha256":"881282db5b5fccf2c72624dbdae2055d310685c7d30c35bc1140ed39fde6707e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:14:45.629671Z","signature_b64":"zSSNvE6MYQ+GWqoFdov/Hwi7SvdL4shvYIgAn8lf8xUVHPxVwWC4TlL4ghvu9SF7rTLW8O60Is7tmega1qmCBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5b29fca42d55334cee01ac335e10718e34fafd351013f0eb83364f13b8fee06","last_reissued_at":"2026-07-05T02:14:45.629229Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:14:45.629229Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PipeTransformer: Automated Elastic Pipelining for Distributed Training of Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chaoyang He, Mahdi Soltanolkotabi, Salman Avestimehr, Shen Li","submitted_at":"2021-02-05T13:39:31Z","abstract_excerpt":"The size of Transformer models is growing at an unprecedented pace. It has only taken less than one year to reach trillion-level parameters after the release of GPT-3 (175B). Training such models requires both substantial engineering efforts and enormous computing resources, which are luxuries most research teams cannot afford. In this paper, we propose PipeTransformer, which leverages automated and elastic pipelining and data parallelism for efficient distributed training of Transformer models. PipeTransformer automatically adjusts the pipelining and data parallelism by identifying and freezi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.03161","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.03161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.03161","created_at":"2026-07-05T02:14:45.629291+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.03161v2","created_at":"2026-07-05T02:14:45.629291+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.03161","created_at":"2026-07-05T02:14:45.629291+00:00"},{"alias_kind":"pith_short_12","alias_value":"UWZJ7SSC2VJT","created_at":"2026-07-05T02:14:45.629291+00:00"},{"alias_kind":"pith_short_16","alias_value":"UWZJ7SSC2VJTJTXA","created_at":"2026-07-05T02:14:45.629291+00:00"},{"alias_kind":"pith_short_8","alias_value":"UWZJ7SSC","created_at":"2026-07-05T02:14:45.629291+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2304.11277","citing_title":"PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD","json":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD.json","graph_json":"https://pith.science/api/pith-number/UWZJ7SSC2VJTJTXADLBTLYIHDD/graph.json","events_json":"https://pith.science/api/pith-number/UWZJ7SSC2VJTJTXADLBTLYIHDD/events.json","paper":"https://pith.science/paper/UWZJ7SSC"},"agent_actions":{"view_html":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD","download_json":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD.json","view_paper":"https://pith.science/paper/UWZJ7SSC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.03161&json=true","fetch_graph":"https://pith.science/api/pith-number/UWZJ7SSC2VJTJTXADLBTLYIHDD/graph.json","fetch_events":"https://pith.science/api/pith-number/UWZJ7SSC2VJTJTXADLBTLYIHDD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD/action/storage_attestation","attest_author":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD/action/author_attestation","sign_citation":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD/action/citation_signature","submit_replication":"https://pith.science/pith/UWZJ7SSC2VJTJTXADLBTLYIHDD/action/replication_record"}},"created_at":"2026-07-05T02:14:45.629291+00:00","updated_at":"2026-07-05T02:14:45.629291+00:00"}