{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3MRRII4V2FCE4PHIWHPV5XAOAZ","short_pith_number":"pith:3MRRII4V","schema_version":"1.0","canonical_sha256":"db23142395d1444e3ce8b1df5edc0e06589e636340651036c49bd4210386b885","source":{"kind":"arxiv","id":"2507.01004","version":2},"attestation_state":"computed","paper":{"title":"ZeCO: Zero Communication Overhead Sequence Parallelism for Linear Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Congying Chu, Jibin Wu, Qian Liu, Ruijie Zhu, Tianjian Li, Xinyi Wan, Yuhong Chou, Zehao Liu, Zejun Ma","submitted_at":"2025-07-01T17:54:53Z","abstract_excerpt":"Linear attention mechanisms deliver significant advantages for Large Language Models (LLMs) by providing linear computational complexity, enabling efficient processing of ultra-long sequences (e.g., 1M context). However, existing Sequence Parallelism (SP) methods, essential for distributing these workloads across devices, become the primary bottleneck due to substantial communication overhead. In this paper, we introduce ZeCO (Zero Communication Overhead) sequence parallelism for linear attention models, a new SP method designed to overcome these limitations and achieve end-to-end near-linear "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.01004","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-01T17:54:53Z","cross_cats_sorted":[],"title_canon_sha256":"6ae572e5a2956d8617cef01232a14730171c90fcab0dfa2bb6043634b3db47d9","abstract_canon_sha256":"28020d36d200f08133b674451cf93d82ccec7178202c2b870bfbd959619f3fe0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:46.413182Z","signature_b64":"DdSGAE5uKhDXcjLiWpacd+YFWRrOpJXuYVq/7oV+/DhgKmWQA4PERDmQs8ySiFvxmNlNkzVEx41LT/Ei8lWnBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db23142395d1444e3ce8b1df5edc0e06589e636340651036c49bd4210386b885","last_reissued_at":"2026-07-05T11:30:46.412643Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:46.412643Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ZeCO: Zero Communication Overhead Sequence Parallelism for Linear Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Congying Chu, Jibin Wu, Qian Liu, Ruijie Zhu, Tianjian Li, Xinyi Wan, Yuhong Chou, Zehao Liu, Zejun Ma","submitted_at":"2025-07-01T17:54:53Z","abstract_excerpt":"Linear attention mechanisms deliver significant advantages for Large Language Models (LLMs) by providing linear computational complexity, enabling efficient processing of ultra-long sequences (e.g., 1M context). However, existing Sequence Parallelism (SP) methods, essential for distributing these workloads across devices, become the primary bottleneck due to substantial communication overhead. In this paper, we introduce ZeCO (Zero Communication Overhead) sequence parallelism for linear attention models, a new SP method designed to overcome these limitations and achieve end-to-end near-linear "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.01004","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.01004/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.01004","created_at":"2026-07-05T11:30:46.412714+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.01004v2","created_at":"2026-07-05T11:30:46.412714+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.01004","created_at":"2026-07-05T11:30:46.412714+00:00"},{"alias_kind":"pith_short_12","alias_value":"3MRRII4V2FCE","created_at":"2026-07-05T11:30:46.412714+00:00"},{"alias_kind":"pith_short_16","alias_value":"3MRRII4V2FCE4PHI","created_at":"2026-07-05T11:30:46.412714+00:00"},{"alias_kind":"pith_short_8","alias_value":"3MRRII4V","created_at":"2026-07-05T11:30:46.412714+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.05276","citing_title":"SpikingBrain: Spiking Brain-inspired Large Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ","json":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ.json","graph_json":"https://pith.science/api/pith-number/3MRRII4V2FCE4PHIWHPV5XAOAZ/graph.json","events_json":"https://pith.science/api/pith-number/3MRRII4V2FCE4PHIWHPV5XAOAZ/events.json","paper":"https://pith.science/paper/3MRRII4V"},"agent_actions":{"view_html":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ","download_json":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ.json","view_paper":"https://pith.science/paper/3MRRII4V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.01004&json=true","fetch_graph":"https://pith.science/api/pith-number/3MRRII4V2FCE4PHIWHPV5XAOAZ/graph.json","fetch_events":"https://pith.science/api/pith-number/3MRRII4V2FCE4PHIWHPV5XAOAZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ/action/storage_attestation","attest_author":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ/action/author_attestation","sign_citation":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ/action/citation_signature","submit_replication":"https://pith.science/pith/3MRRII4V2FCE4PHIWHPV5XAOAZ/action/replication_record"}},"created_at":"2026-07-05T11:30:46.412714+00:00","updated_at":"2026-07-05T11:30:46.412714+00:00"}