{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GIGJALGBGRNFAAXAM7L4DOJFR2","short_pith_number":"pith:GIGJALGB","schema_version":"1.0","canonical_sha256":"320c902cc1345a5002e067d7c1b9258ea841ab0605e6c5c787178d6e9d1da64d","source":{"kind":"arxiv","id":"2312.17044","version":5},"attestation_state":"computed","paper":{"title":"Length Extrapolation of Transformers: A Survey from the Perspective of Positional Encoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Qin, Dongliang Xu, Hongtao Liu, Liang Zhao, Qing Yang, Ting Liu, Weihong Zhong, Xiachong Feng, Xiaocheng Feng","submitted_at":"2023-12-28T14:42:24Z","abstract_excerpt":"Built upon the Transformer, large language models (LLMs) have captured worldwide attention due to their remarkable abilities. Nevertheless, all Transformer-based models including LLMs suffer from a preset length limit and can hardly generalize from short training sequences to longer inference ones, namely, they cannot perform length extrapolation to handle long sequences, which severely hinders their application in scenarios demanding long input sequences such as legal or scientific documents. Thus, numerous methods have emerged to enhance the length extrapolation of Transformers. Despite the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.17044","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-28T14:42:24Z","cross_cats_sorted":[],"title_canon_sha256":"038a0ff3e6060041d70e3735d1845a8d0e8b4664c517a2f9fe6b33517d734a81","abstract_canon_sha256":"d1d5ca054b8d7fb336bec718de72be13334cd6cdfe2fba3d51f405e63185e4a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:11.197567Z","signature_b64":"gFIZlru6eATZRWW03JMEu+5ZdwmtzWw2/t+QoA+LJK0c2FUfK8iCMxaAIJrwkXH78SCx84o6HehMXIedS/0bBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"320c902cc1345a5002e067d7c1b9258ea841ab0605e6c5c787178d6e9d1da64d","last_reissued_at":"2026-07-05T09:16:11.197079Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:11.197079Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Length Extrapolation of Transformers: A Survey from the Perspective of Positional Encoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Qin, Dongliang Xu, Hongtao Liu, Liang Zhao, Qing Yang, Ting Liu, Weihong Zhong, Xiachong Feng, Xiaocheng Feng","submitted_at":"2023-12-28T14:42:24Z","abstract_excerpt":"Built upon the Transformer, large language models (LLMs) have captured worldwide attention due to their remarkable abilities. Nevertheless, all Transformer-based models including LLMs suffer from a preset length limit and can hardly generalize from short training sequences to longer inference ones, namely, they cannot perform length extrapolation to handle long sequences, which severely hinders their application in scenarios demanding long input sequences such as legal or scientific documents. Thus, numerous methods have emerged to enhance the length extrapolation of Transformers. Despite the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.17044","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.17044/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.17044","created_at":"2026-07-05T09:16:11.197129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.17044v5","created_at":"2026-07-05T09:16:11.197129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.17044","created_at":"2026-07-05T09:16:11.197129+00:00"},{"alias_kind":"pith_short_12","alias_value":"GIGJALGBGRNF","created_at":"2026-07-05T09:16:11.197129+00:00"},{"alias_kind":"pith_short_16","alias_value":"GIGJALGBGRNFAAXA","created_at":"2026-07-05T09:16:11.197129+00:00"},{"alias_kind":"pith_short_8","alias_value":"GIGJALGB","created_at":"2026-07-05T09:16:11.197129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.12370","citing_title":"Positional Encoding in Transformer-Based Time Series Models: A Survey","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2504.14386","citing_title":"LOOPE: Learnable Optimal Patch Order in Positional Embeddings for Vision Transformers","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19944","citing_title":"A Measure-Theoretic Analysis of Reasoning: Structural Generalization and Approximation Limits","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.04154","citing_title":"Robust Filter Attention: Self-Attention as Precision-Weighted State Estimation","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01651","citing_title":"On the Spatiotemporal Dynamics of Generalization in Neural Networks","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29069","citing_title":"On the Mirage of Long-Range Dependency, with an Application to Integer Multiplication","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2","json":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2.json","graph_json":"https://pith.science/api/pith-number/GIGJALGBGRNFAAXAM7L4DOJFR2/graph.json","events_json":"https://pith.science/api/pith-number/GIGJALGBGRNFAAXAM7L4DOJFR2/events.json","paper":"https://pith.science/paper/GIGJALGB"},"agent_actions":{"view_html":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2","download_json":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2.json","view_paper":"https://pith.science/paper/GIGJALGB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.17044&json=true","fetch_graph":"https://pith.science/api/pith-number/GIGJALGBGRNFAAXAM7L4DOJFR2/graph.json","fetch_events":"https://pith.science/api/pith-number/GIGJALGBGRNFAAXAM7L4DOJFR2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2/action/storage_attestation","attest_author":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2/action/author_attestation","sign_citation":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2/action/citation_signature","submit_replication":"https://pith.science/pith/GIGJALGBGRNFAAXAM7L4DOJFR2/action/replication_record"}},"created_at":"2026-07-05T09:16:11.197129+00:00","updated_at":"2026-07-05T09:16:11.197129+00:00"}