{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:T465Z3ZGD2PPIS4G7U6L2NSGNW","short_pith_number":"pith:T465Z3ZG","schema_version":"1.0","canonical_sha256":"9f3ddcef261e9ef44b86fd3cbd36466d9cd98a765ec7fe63f8c9eac16a7fa888","source":{"kind":"arxiv","id":"2211.13878","version":1},"attestation_state":"computed","paper":{"title":"Galvatron: Efficient Transformer Training over Multiple GPUs Using Automatic Parallelism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DB","cs.DC"],"primary_cat":"cs.LG","authors_text":"Bin Cui, Chunan Shi, Hailin Zhang, Xiaonan Nie, Xupeng Miao, Youhe Jiang, Yujie Wang","submitted_at":"2022-11-25T03:45:31Z","abstract_excerpt":"Transformer models have achieved state-of-the-art performance on various domains of applications and gradually becomes the foundations of the advanced large deep learning (DL) models. However, how to train these models over multiple GPUs efficiently is still challenging due to a large number of parallelism choices. Existing DL systems either rely on manual efforts to make distributed training plans or apply parallelism combinations within a very limited search space. In this approach, we propose Galvatron, a new system framework that incorporates multiple popular parallelism dimensions and aut"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.13878","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-25T03:45:31Z","cross_cats_sorted":["cs.DB","cs.DC"],"title_canon_sha256":"3d833bd157113665f11023d3d5ab87046635aa9dd331356f93859b059a8591c5","abstract_canon_sha256":"4cb796705831551840479bb02b596ae440a1f03e9c0e45851d735c51075dc974"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:19:27.326784Z","signature_b64":"nPxdTml56Yzcro7Acqad3uZr0erQ+7x16QyZJCqharvvTJlRXZo+S+cdFfoLEtnBjyua+ei7fOOaokcrq8BPBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f3ddcef261e9ef44b86fd3cbd36466d9cd98a765ec7fe63f8c9eac16a7fa888","last_reissued_at":"2026-07-05T05:19:27.326284Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:19:27.326284Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Galvatron: Efficient Transformer Training over Multiple GPUs Using Automatic Parallelism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DB","cs.DC"],"primary_cat":"cs.LG","authors_text":"Bin Cui, Chunan Shi, Hailin Zhang, Xiaonan Nie, Xupeng Miao, Youhe Jiang, Yujie Wang","submitted_at":"2022-11-25T03:45:31Z","abstract_excerpt":"Transformer models have achieved state-of-the-art performance on various domains of applications and gradually becomes the foundations of the advanced large deep learning (DL) models. However, how to train these models over multiple GPUs efficiently is still challenging due to a large number of parallelism choices. Existing DL systems either rely on manual efforts to make distributed training plans or apply parallelism combinations within a very limited search space. In this approach, we propose Galvatron, a new system framework that incorporates multiple popular parallelism dimensions and aut"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.13878","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.13878/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.13878","created_at":"2026-07-05T05:19:27.326348+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.13878v1","created_at":"2026-07-05T05:19:27.326348+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.13878","created_at":"2026-07-05T05:19:27.326348+00:00"},{"alias_kind":"pith_short_12","alias_value":"T465Z3ZGD2PP","created_at":"2026-07-05T05:19:27.326348+00:00"},{"alias_kind":"pith_short_16","alias_value":"T465Z3ZGD2PPIS4G","created_at":"2026-07-05T05:19:27.326348+00:00"},{"alias_kind":"pith_short_8","alias_value":"T465Z3ZG","created_at":"2026-07-05T05:19:27.326348+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23087","citing_title":"FlowTrain: Flow-Based Decoupled Training for Industrial-Grade Vision-Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11169","citing_title":"Piper: A Programmable Distributed Training System","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22014","citing_title":"LiveR: Fine-Grained Elasticity via Live Reconfiguration for Model Training","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21312","citing_title":"Frontier: Towards Comprehensive and Accurate LLM Inference Simulation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16637","citing_title":"HexAGenT: Efficient Agentic LLM Serving via Workflow- and Heterogeneity-Aware Scheduling","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19654","citing_title":"FEPLB: Exploiting Copy Engines for Nearly Free MoE Load Balancing in Distributed Training","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07144","citing_title":"Autopoiesis: A Self-Evolving System Paradigm for LLM Serving Under Runtime Dynamics","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07569","citing_title":"HexiSeq: Accommodating Long Context Training of LLMs over Heterogeneous Hardware","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW","json":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW.json","graph_json":"https://pith.science/api/pith-number/T465Z3ZGD2PPIS4G7U6L2NSGNW/graph.json","events_json":"https://pith.science/api/pith-number/T465Z3ZGD2PPIS4G7U6L2NSGNW/events.json","paper":"https://pith.science/paper/T465Z3ZG"},"agent_actions":{"view_html":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW","download_json":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW.json","view_paper":"https://pith.science/paper/T465Z3ZG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.13878&json=true","fetch_graph":"https://pith.science/api/pith-number/T465Z3ZGD2PPIS4G7U6L2NSGNW/graph.json","fetch_events":"https://pith.science/api/pith-number/T465Z3ZGD2PPIS4G7U6L2NSGNW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW/action/storage_attestation","attest_author":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW/action/author_attestation","sign_citation":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW/action/citation_signature","submit_replication":"https://pith.science/pith/T465Z3ZGD2PPIS4G7U6L2NSGNW/action/replication_record"}},"created_at":"2026-07-05T05:19:27.326348+00:00","updated_at":"2026-07-05T05:19:27.326348+00:00"}