{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EHLTMDWH7ZLJUWMQLFY6WADPHY","short_pith_number":"pith:EHLTMDWH","schema_version":"1.0","canonical_sha256":"21d7360ec7fe569a59905971eb006f3e13e472d2d521106f9547280c58e57ef8","source":{"kind":"arxiv","id":"2405.15780","version":1},"attestation_state":"computed","paper":{"title":"Sequence Length Scaling in Vision Transformers for Scientific Images on Frontier","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Aristeidis Tsaris, Chengming Zhang, Dan Lu, Feiyi Wang, Jong Youl Choi, Junqi Yin, Ming Fan, Moetasim Ashfaq, Mohamed Wahib, Prasanna Balaprakash, Siyan Liu, Xiao Wang","submitted_at":"2024-04-17T19:57:07Z","abstract_excerpt":"Vision Transformers (ViTs) are pivotal for foundational models in scientific imagery, including Earth science applications, due to their capability to process large sequence lengths. While transformers for text has inspired scaling sequence lengths in ViTs, yet adapting these for ViTs introduces unique challenges. We develop distributed sequence parallelism for ViTs, enabling them to handle up to 1M tokens. Our approach, leveraging DeepSpeed-Ulysses and Long-Sequence-Segmentation with model sharding, is the first to apply sequence parallelism in ViT training, achieving a 94% batch scaling effi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15780","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-17T19:57:07Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a53a88831654a05ca6ef85082214577a9eba6407b8297b91944567c8c0f65e0d","abstract_canon_sha256":"4844524fc8a30006e498e5d5848aaa1b5b704b1a44213dbbe5962b1623e7ff88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:56.484411Z","signature_b64":"Bo5pcsiF7dUQnROyXPGLhc6E+nyNA3LON90i44KqqyWk06EPgZkkGy6mDBzOd8p+KfaO9HW0I5JgC9fPq948CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"21d7360ec7fe569a59905971eb006f3e13e472d2d521106f9547280c58e57ef8","last_reissued_at":"2026-07-05T08:22:56.483911Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:56.483911Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sequence Length Scaling in Vision Transformers for Scientific Images on Frontier","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Aristeidis Tsaris, Chengming Zhang, Dan Lu, Feiyi Wang, Jong Youl Choi, Junqi Yin, Ming Fan, Moetasim Ashfaq, Mohamed Wahib, Prasanna Balaprakash, Siyan Liu, Xiao Wang","submitted_at":"2024-04-17T19:57:07Z","abstract_excerpt":"Vision Transformers (ViTs) are pivotal for foundational models in scientific imagery, including Earth science applications, due to their capability to process large sequence lengths. While transformers for text has inspired scaling sequence lengths in ViTs, yet adapting these for ViTs introduces unique challenges. We develop distributed sequence parallelism for ViTs, enabling them to handle up to 1M tokens. Our approach, leveraging DeepSpeed-Ulysses and Long-Sequence-Segmentation with model sharding, is the first to apply sequence parallelism in ViT training, achieving a 94% batch scaling effi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15780","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15780/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15780","created_at":"2026-07-05T08:22:56.483971+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15780v1","created_at":"2026-07-05T08:22:56.483971+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15780","created_at":"2026-07-05T08:22:56.483971+00:00"},{"alias_kind":"pith_short_12","alias_value":"EHLTMDWH7ZLJ","created_at":"2026-07-05T08:22:56.483971+00:00"},{"alias_kind":"pith_short_16","alias_value":"EHLTMDWH7ZLJUWMQ","created_at":"2026-07-05T08:22:56.483971+00:00"},{"alias_kind":"pith_short_8","alias_value":"EHLTMDWH","created_at":"2026-07-05T08:22:56.483971+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.04802","citing_title":"ORBIT-2: Scaling Exascale Vision Foundation Models for Weather and Climate Downscaling","ref_index":40,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY","json":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY.json","graph_json":"https://pith.science/api/pith-number/EHLTMDWH7ZLJUWMQLFY6WADPHY/graph.json","events_json":"https://pith.science/api/pith-number/EHLTMDWH7ZLJUWMQLFY6WADPHY/events.json","paper":"https://pith.science/paper/EHLTMDWH"},"agent_actions":{"view_html":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY","download_json":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY.json","view_paper":"https://pith.science/paper/EHLTMDWH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15780&json=true","fetch_graph":"https://pith.science/api/pith-number/EHLTMDWH7ZLJUWMQLFY6WADPHY/graph.json","fetch_events":"https://pith.science/api/pith-number/EHLTMDWH7ZLJUWMQLFY6WADPHY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY/action/storage_attestation","attest_author":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY/action/author_attestation","sign_citation":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY/action/citation_signature","submit_replication":"https://pith.science/pith/EHLTMDWH7ZLJUWMQLFY6WADPHY/action/replication_record"}},"created_at":"2026-07-05T08:22:56.483971+00:00","updated_at":"2026-07-05T08:22:56.483971+00:00"}