{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:DZ2DMKHAHAH4MQIDYGRIIT4RTV","short_pith_number":"pith:DZ2DMKHA","schema_version":"1.0","canonical_sha256":"1e743628e0380fc64103c1a2844f919d60007e02b7864d1eefdaad6a0657f0d4","source":{"kind":"arxiv","id":"2105.13120","version":3},"attestation_state":"computed","paper":{"title":"Sequence Parallelism: Long Sequence Training from System Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Chaitanya Baranwal, Fuzhao Xue, Shenggui Li, Yang You, Yongbin Li","submitted_at":"2021-05-26T13:40:58Z","abstract_excerpt":"Transformer achieves promising results on various tasks. However, self-attention suffers from quadratic memory requirements with respect to the sequence length. Existing work focuses on reducing time and space complexity from an algorithm perspective. In this work, we propose sequence parallelism, a memory-efficient parallelism method to help us break input sequence length limitation and train with longer sequences on GPUs efficiently. Our approach is compatible with most existing parallelisms (e.g. data parallelism, pipeline parallelism and tensor parallelism), which means our sequence parall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.13120","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-26T13:40:58Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"18a3b110f4fa3f62c52cbfc5bced3a87e4e240622be1634a4d077fc3e5b9c0ff","abstract_canon_sha256":"75be1b1ac93979806cbd0a43146cffa19616bf8ae093159e2afe647cfa2266e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:25:10.267645Z","signature_b64":"jMm4v3TqjCcz7coLnz+ep5ba0dNOmO/e7K88n1h1i6g+jdop2oJOWU+oj93ldUgIcmmRn/7DgfbTtvQfu5ovBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e743628e0380fc64103c1a2844f919d60007e02b7864d1eefdaad6a0657f0d4","last_reissued_at":"2026-07-05T04:25:10.267213Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:25:10.267213Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sequence Parallelism: Long Sequence Training from System Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Chaitanya Baranwal, Fuzhao Xue, Shenggui Li, Yang You, Yongbin Li","submitted_at":"2021-05-26T13:40:58Z","abstract_excerpt":"Transformer achieves promising results on various tasks. However, self-attention suffers from quadratic memory requirements with respect to the sequence length. Existing work focuses on reducing time and space complexity from an algorithm perspective. In this work, we propose sequence parallelism, a memory-efficient parallelism method to help us break input sequence length limitation and train with longer sequences on GPUs efficiently. Our approach is compatible with most existing parallelisms (e.g. data parallelism, pipeline parallelism and tensor parallelism), which means our sequence parall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.13120","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.13120/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.13120","created_at":"2026-07-05T04:25:10.267269+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.13120v3","created_at":"2026-07-05T04:25:10.267269+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.13120","created_at":"2026-07-05T04:25:10.267269+00:00"},{"alias_kind":"pith_short_12","alias_value":"DZ2DMKHAHAH4","created_at":"2026-07-05T04:25:10.267269+00:00"},{"alias_kind":"pith_short_16","alias_value":"DZ2DMKHAHAH4MQID","created_at":"2026-07-05T04:25:10.267269+00:00"},{"alias_kind":"pith_short_8","alias_value":"DZ2DMKHA","created_at":"2026-07-05T04:25:10.267269+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22541","citing_title":"ASAP: A Disaggregated and Asynchronous Inference System for MoE Prefill","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19989","citing_title":"Online Dynamic Batching with Formal Guarantees for LLM Training","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01646","citing_title":"PHOENIX: Resilient LLM Training with Hot-Swapping via Zero-Overhead Checkpoint","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01817","citing_title":"HCMS: Head-Chunked Multi-Stream Pipeline for Communication-Computation Overlap in Long-Sequence Parallel Attention","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17164","citing_title":"Charon: A Unified and Fine-Grained Simulator for Large-Scale LLM Training and Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18710","citing_title":"Mosaic: Towards Efficient Training of Multimodal Models with Spatial Resource Multiplexing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16867","citing_title":"The Falcon Series of Open Language Models","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2402.08268","citing_title":"World Model on Million-Length Video And Language With Blockwise RingAttention","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2403.04652","citing_title":"Yi: Open Foundation Models by 01.AI","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05628","citing_title":"Towards Compute-Aware In-Switch Computing for LLMs Tensor-Parallelism on Multi-GPU Systems","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV","json":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV.json","graph_json":"https://pith.science/api/pith-number/DZ2DMKHAHAH4MQIDYGRIIT4RTV/graph.json","events_json":"https://pith.science/api/pith-number/DZ2DMKHAHAH4MQIDYGRIIT4RTV/events.json","paper":"https://pith.science/paper/DZ2DMKHA"},"agent_actions":{"view_html":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV","download_json":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV.json","view_paper":"https://pith.science/paper/DZ2DMKHA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.13120&json=true","fetch_graph":"https://pith.science/api/pith-number/DZ2DMKHAHAH4MQIDYGRIIT4RTV/graph.json","fetch_events":"https://pith.science/api/pith-number/DZ2DMKHAHAH4MQIDYGRIIT4RTV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV/action/storage_attestation","attest_author":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV/action/author_attestation","sign_citation":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV/action/citation_signature","submit_replication":"https://pith.science/pith/DZ2DMKHAHAH4MQIDYGRIIT4RTV/action/replication_record"}},"created_at":"2026-07-05T04:25:10.267269+00:00","updated_at":"2026-07-05T04:25:10.267269+00:00"}