{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XF35BV7U2JIAANYZUYN66CNCMF","short_pith_number":"pith:XF35BV7U","schema_version":"1.0","canonical_sha256":"b977d0d7f4d250003719a61bef09a2615dafa6e6fa8827c35f36f725957a3a31","source":{"kind":"arxiv","id":"2404.12224","version":2},"attestation_state":"computed","paper":{"title":"Length Generalization of Causal Transformers without Position Encoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Yan, Jie Wang, Qi Zhang, Tao Gui, Tao Ji, Xiaoling Wang, Xuanjing Huang, Yuanbin Wu","submitted_at":"2024-04-18T14:38:32Z","abstract_excerpt":"Generalizing to longer sentences is important for recent Transformer-based language models. Besides algorithms manipulating explicit position features, the success of Transformers without position encodings (NoPE) provides a new way to overcome the challenge. In this paper, we study the length generalization property of NoPE. We find that although NoPE can extend to longer sequences than the commonly used explicit position encodings, it still has a limited context length. We identify a connection between the failure of NoPE's generalization and the distraction of attention distributions. We pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.12224","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-18T14:38:32Z","cross_cats_sorted":[],"title_canon_sha256":"9a494823dd6b09622a71bfc1fa9f4f403abd18896a84a8c9c773a9c26171fc71","abstract_canon_sha256":"ae3cabf0986a8b14020e34ded3a94aa75a581fcc1d92e53e50706e0319bcf6b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:04.654155Z","signature_b64":"UlSE7Lt+kuK9RtOq8EqXKpd0XDsH/qYT3lNH2t4wzCT4wfW386nz+TH2Tr3AHs5D+r0V6spOx/mNiNUkSpm1Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b977d0d7f4d250003719a61bef09a2615dafa6e6fa8827c35f36f725957a3a31","last_reissued_at":"2026-07-05T08:24:04.653708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:04.653708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Length Generalization of Causal Transformers without Position Encoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Yan, Jie Wang, Qi Zhang, Tao Gui, Tao Ji, Xiaoling Wang, Xuanjing Huang, Yuanbin Wu","submitted_at":"2024-04-18T14:38:32Z","abstract_excerpt":"Generalizing to longer sentences is important for recent Transformer-based language models. Besides algorithms manipulating explicit position features, the success of Transformers without position encodings (NoPE) provides a new way to overcome the challenge. In this paper, we study the length generalization property of NoPE. We find that although NoPE can extend to longer sequences than the commonly used explicit position encodings, it still has a limited context length. We identify a connection between the failure of NoPE's generalization and the distraction of attention distributions. We pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12224","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12224/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.12224","created_at":"2026-07-05T08:24:04.653773+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.12224v2","created_at":"2026-07-05T08:24:04.653773+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12224","created_at":"2026-07-05T08:24:04.653773+00:00"},{"alias_kind":"pith_short_12","alias_value":"XF35BV7U2JIA","created_at":"2026-07-05T08:24:04.653773+00:00"},{"alias_kind":"pith_short_16","alias_value":"XF35BV7U2JIAANYZ","created_at":"2026-07-05T08:24:04.653773+00:00"},{"alias_kind":"pith_short_8","alias_value":"XF35BV7U","created_at":"2026-07-05T08:24:04.653773+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25156","citing_title":"ATMA: Length-Invariant Language Modeling via Polar Attention and Gated-Delta Compression Memory","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25156","citing_title":"ATMA: Length-Invariant Language Modeling via Polar Attention and Gated-Delta Compression Memory","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18603","citing_title":"Dual Triangle Attention: Effective Bidirectional Attention Without Positional Embeddings","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF","json":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF.json","graph_json":"https://pith.science/api/pith-number/XF35BV7U2JIAANYZUYN66CNCMF/graph.json","events_json":"https://pith.science/api/pith-number/XF35BV7U2JIAANYZUYN66CNCMF/events.json","paper":"https://pith.science/paper/XF35BV7U"},"agent_actions":{"view_html":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF","download_json":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF.json","view_paper":"https://pith.science/paper/XF35BV7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.12224&json=true","fetch_graph":"https://pith.science/api/pith-number/XF35BV7U2JIAANYZUYN66CNCMF/graph.json","fetch_events":"https://pith.science/api/pith-number/XF35BV7U2JIAANYZUYN66CNCMF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF/action/storage_attestation","attest_author":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF/action/author_attestation","sign_citation":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF/action/citation_signature","submit_replication":"https://pith.science/pith/XF35BV7U2JIAANYZUYN66CNCMF/action/replication_record"}},"created_at":"2026-07-05T08:24:04.653773+00:00","updated_at":"2026-07-05T08:24:04.653773+00:00"}