{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4XOFMGAHR4DBVUHY7AKNOU3BRZ","short_pith_number":"pith:4XOFMGAH","schema_version":"1.0","canonical_sha256":"e5dc5618078f061ad0f8f814d753618e4bd638ff8116e0497b69ac69f3a670dc","source":{"kind":"arxiv","id":"2110.05722","version":3},"attestation_state":"computed","paper":{"title":"LightSeq2: Accelerated Training for Transformer-based Models on GPUs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MS"],"primary_cat":"cs.CL","authors_text":"Guyue Huang, Lei Li, Mingxuan Wang, Xian Qian, Xiaohui Wang, Yang Wei, Ying Xiong, Yufei Ding","submitted_at":"2021-10-12T03:17:03Z","abstract_excerpt":"Transformer-based neural models are used in many AI applications. Training these models is expensive, as it takes huge GPU resources and long duration. It is challenging because typical data like sentences have variable lengths, and Transformer's computation patterns are more complex than convolutional neural networks. Existing systems either only focus on model inference or optimization for only BERT-like encoder models. In this paper, we present LightSeq2, a system to accelerate training for a general family of Transformer models on GPUs. We propose a series of GPU optimization techniques ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.05722","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-10-12T03:17:03Z","cross_cats_sorted":["cs.MS"],"title_canon_sha256":"9e3ffeefa43dfe1b243dd565903af4a4078fb11cde24c811196c78a9178de7e8","abstract_canon_sha256":"4fc1588fce2d0695d28c2e2a08e70263ff56ffbbebcce1f6cb7b177975f9db34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:32:12.289344Z","signature_b64":"kCb9dKOiZPWQK9Xrl2p1NOtuJ1L6BpxqBkDF75t71f/a4ZK3E/YOpe/1tvEfakCWorqXxPISKYRPWibE6zS/CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5dc5618078f061ad0f8f814d753618e4bd638ff8116e0497b69ac69f3a670dc","last_reissued_at":"2026-07-05T04:32:12.288841Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:32:12.288841Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LightSeq2: Accelerated Training for Transformer-based Models on GPUs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MS"],"primary_cat":"cs.CL","authors_text":"Guyue Huang, Lei Li, Mingxuan Wang, Xian Qian, Xiaohui Wang, Yang Wei, Ying Xiong, Yufei Ding","submitted_at":"2021-10-12T03:17:03Z","abstract_excerpt":"Transformer-based neural models are used in many AI applications. Training these models is expensive, as it takes huge GPU resources and long duration. It is challenging because typical data like sentences have variable lengths, and Transformer's computation patterns are more complex than convolutional neural networks. Existing systems either only focus on model inference or optimization for only BERT-like encoder models. In this paper, we present LightSeq2, a system to accelerate training for a general family of Transformer models on GPUs. We propose a series of GPU optimization techniques ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.05722","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.05722/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.05722","created_at":"2026-07-05T04:32:12.288900+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.05722v3","created_at":"2026-07-05T04:32:12.288900+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.05722","created_at":"2026-07-05T04:32:12.288900+00:00"},{"alias_kind":"pith_short_12","alias_value":"4XOFMGAHR4DB","created_at":"2026-07-05T04:32:12.288900+00:00"},{"alias_kind":"pith_short_16","alias_value":"4XOFMGAHR4DBVUHY","created_at":"2026-07-05T04:32:12.288900+00:00"},{"alias_kind":"pith_short_8","alias_value":"4XOFMGAH","created_at":"2026-07-05T04:32:12.288900+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ","json":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ.json","graph_json":"https://pith.science/api/pith-number/4XOFMGAHR4DBVUHY7AKNOU3BRZ/graph.json","events_json":"https://pith.science/api/pith-number/4XOFMGAHR4DBVUHY7AKNOU3BRZ/events.json","paper":"https://pith.science/paper/4XOFMGAH"},"agent_actions":{"view_html":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ","download_json":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ.json","view_paper":"https://pith.science/paper/4XOFMGAH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.05722&json=true","fetch_graph":"https://pith.science/api/pith-number/4XOFMGAHR4DBVUHY7AKNOU3BRZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4XOFMGAHR4DBVUHY7AKNOU3BRZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ/action/storage_attestation","attest_author":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ/action/author_attestation","sign_citation":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ/action/citation_signature","submit_replication":"https://pith.science/pith/4XOFMGAHR4DBVUHY7AKNOU3BRZ/action/replication_record"}},"created_at":"2026-07-05T04:32:12.288900+00:00","updated_at":"2026-07-05T04:32:12.288900+00:00"}