{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FV6BMPRKZQFG4IIMKUDHVYKNSF","short_pith_number":"pith:FV6BMPRK","schema_version":"1.0","canonical_sha256":"2d7c163e2acc0a6e210c55067ae14d91786fb7d56bf1f5db240843dc58a52eb4","source":{"kind":"arxiv","id":"2307.02486","version":2},"attestation_state":"computed","paper":{"title":"LongNet: Scaling Transformers to 1,000,000,000 Tokens","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Jiayu Ding, Li Dong, Nanning Zheng, Shaohan Huang, Shuming Ma, Wenhui Wang, Xingxing Zhang","submitted_at":"2023-07-05T17:59:38Z","abstract_excerpt":"Scaling sequence length has become a critical demand in the era of large language models. However, existing methods struggle with either computational complexity or model expressivity, rendering the maximum sequence length restricted. To address this issue, we introduce LongNet, a Transformer variant that can scale sequence length to more than 1 billion tokens, without sacrificing the performance on shorter sequences. Specifically, we propose dilated attention, which expands the attentive field exponentially as the distance grows. LongNet has significant advantages: 1) it has a linear computat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.02486","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-05T17:59:38Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1396a5771b9f0c94afb9df6cb140ad2bcc9cd0568e671236fdc01f20159c70bd","abstract_canon_sha256":"6721b45842cea923d7b18d4eea3082d03b8da895b61faccfe72cd53783f91118"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:32:40.228607Z","signature_b64":"IvBJ1U/z/FX6CKa4QJTXXik6DIDQYWiodg9R+bsB+uifIb5Gv5R/sKaG5mXQulAiUVpziVsyfLMvKN1mvf5BDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d7c163e2acc0a6e210c55067ae14d91786fb7d56bf1f5db240843dc58a52eb4","last_reissued_at":"2026-07-05T06:32:40.228099Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:32:40.228099Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongNet: Scaling Transformers to 1,000,000,000 Tokens","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Jiayu Ding, Li Dong, Nanning Zheng, Shaohan Huang, Shuming Ma, Wenhui Wang, Xingxing Zhang","submitted_at":"2023-07-05T17:59:38Z","abstract_excerpt":"Scaling sequence length has become a critical demand in the era of large language models. However, existing methods struggle with either computational complexity or model expressivity, rendering the maximum sequence length restricted. To address this issue, we introduce LongNet, a Transformer variant that can scale sequence length to more than 1 billion tokens, without sacrificing the performance on shorter sequences. Specifically, we propose dilated attention, which expands the attentive field exponentially as the distance grows. LongNet has significant advantages: 1) it has a linear computat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.02486","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.02486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.02486","created_at":"2026-07-05T06:32:40.228158+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.02486v2","created_at":"2026-07-05T06:32:40.228158+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.02486","created_at":"2026-07-05T06:32:40.228158+00:00"},{"alias_kind":"pith_short_12","alias_value":"FV6BMPRKZQFG","created_at":"2026-07-05T06:32:40.228158+00:00"},{"alias_kind":"pith_short_16","alias_value":"FV6BMPRKZQFG4IIM","created_at":"2026-07-05T06:32:40.228158+00:00"},{"alias_kind":"pith_short_8","alias_value":"FV6BMPRK","created_at":"2026-07-05T06:32:40.228158+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24331","citing_title":"Transformer-Based Language Models Across Domain Verticals: Architectures, Applications and Critical Assessment","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06453","citing_title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02680","citing_title":"Locality Does Not Imply Reachability: Boundary Repair in Block-Sparse Causal Attention","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28560","citing_title":"Depth-Staggered Fibonacci Spacing for Sparse Attention: Static Schedules Beat Learned Dilation and Extrapolate Where Dense Attention Fails","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29844","citing_title":"MATCH: Modulating Attention via In-Context Retrieval for Long-Context Transformers","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30716","citing_title":"Simple Token-Efficient Vision-Language Model for Case-level Pathology Synoptic Report Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13368","citing_title":"BlossomRec: Block-level Fused Sparse Attention Mechanism for Sequential Recommendations","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04960","citing_title":"On Efficient Variants of Segment Anything Model: A Survey","ref_index":189,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07143","citing_title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18196","citing_title":"RAT+: Train Dense, Infer Sparse -- Recurrence Augmented Attention for Dilated Inference","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06313","citing_title":"Beyond Chunking: Discourse-Aware Hierarchical Retrieval for Long Document Question Answering","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15155","citing_title":"eLLM: Elastic Memory Management Framework for Efficient LLM Serving","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21526","citing_title":"Accelerating Prefilling via Decoding-time Contribution Sparsity","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12635","citing_title":"Positional Encoding via Token-Aware Phase Attention","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12602","citing_title":"Exact Flow Linear Attention: Exact Solution from Continuous-Time Dynamics","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18196","citing_title":"RAT+: Train Dense, Infer Sparse -- Recurrence Augmented Attention for Dilated Inference","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04759","citing_title":"Stacked from One: Multi-Scale Self-Injection for Context Window Extension","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2308.14508","citing_title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10544","citing_title":"Where Does Long-Context Supervision Actually Go? Effective-Context Exposure Balancing","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2401.09417","citing_title":"Vision Mamba: Efficient Visual Representation Learning with Bidirectional State Space Model","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2404.06654","citing_title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2312.00752","citing_title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF","json":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF.json","graph_json":"https://pith.science/api/pith-number/FV6BMPRKZQFG4IIMKUDHVYKNSF/graph.json","events_json":"https://pith.science/api/pith-number/FV6BMPRKZQFG4IIMKUDHVYKNSF/events.json","paper":"https://pith.science/paper/FV6BMPRK"},"agent_actions":{"view_html":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF","download_json":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF.json","view_paper":"https://pith.science/paper/FV6BMPRK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.02486&json=true","fetch_graph":"https://pith.science/api/pith-number/FV6BMPRKZQFG4IIMKUDHVYKNSF/graph.json","fetch_events":"https://pith.science/api/pith-number/FV6BMPRKZQFG4IIMKUDHVYKNSF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF/action/storage_attestation","attest_author":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF/action/author_attestation","sign_citation":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF/action/citation_signature","submit_replication":"https://pith.science/pith/FV6BMPRKZQFG4IIMKUDHVYKNSF/action/replication_record"}},"created_at":"2026-07-05T06:32:40.228158+00:00","updated_at":"2026-07-05T06:32:40.228158+00:00"}