{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3Y7RFGYFG6WPQWASZPC3F3ZH2G","short_pith_number":"pith:3Y7RFGYF","schema_version":"1.0","canonical_sha256":"de3f129b0537acf85812cbc5b2ef27d1b6160838d0be592f028f39bec0c42c8b","source":{"kind":"arxiv","id":"2410.18745","version":1},"attestation_state":"computed","paper":{"title":"Why Does the Effective Context Length of LLMs Fall Short?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenxin An, Jingjing Xu, Jun Zhang, Lei Li, Lingpeng Kong, Ming Zhong, Shansan Gong, Yao Luo","submitted_at":"2024-10-24T13:51:50Z","abstract_excerpt":"Advancements in distributed training and efficient attention mechanisms have significantly expanded the context window sizes of large language models (LLMs). However, recent work reveals that the effective context lengths of open-source LLMs often fall short, typically not exceeding half of their training lengths. In this work, we attribute this limitation to the left-skewed frequency distribution of relative positions formed in LLMs pretraining and post-training stages, which impedes their ability to effectively gather distant information. To address this challenge, we introduce ShifTed Rotra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18745","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T13:51:50Z","cross_cats_sorted":[],"title_canon_sha256":"db2db5f47504cbba0a26563703c091bb0ad6dd2e3a8028c6208be33d3635f87a","abstract_canon_sha256":"ab19366d235e516c642169433c6c0f91f9437ef2ab82685d0141abe1ee540d6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:12.996164Z","signature_b64":"CdfDQfIdkeGhJ7Nn56lVe3/2wwSgr3rDRdUFy1+dbIgqUiEUntMDv2ToJTcfTbeXi3RH+R4tXp3sk9l/VLdcCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de3f129b0537acf85812cbc5b2ef27d1b6160838d0be592f028f39bec0c42c8b","last_reissued_at":"2026-07-05T09:25:12.995654Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:12.995654Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Does the Effective Context Length of LLMs Fall Short?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenxin An, Jingjing Xu, Jun Zhang, Lei Li, Lingpeng Kong, Ming Zhong, Shansan Gong, Yao Luo","submitted_at":"2024-10-24T13:51:50Z","abstract_excerpt":"Advancements in distributed training and efficient attention mechanisms have significantly expanded the context window sizes of large language models (LLMs). However, recent work reveals that the effective context lengths of open-source LLMs often fall short, typically not exceeding half of their training lengths. In this work, we attribute this limitation to the left-skewed frequency distribution of relative positions formed in LLMs pretraining and post-training stages, which impedes their ability to effectively gather distant information. To address this challenge, we introduce ShifTed Rotra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18745","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18745/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18745","created_at":"2026-07-05T09:25:12.995713+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18745v1","created_at":"2026-07-05T09:25:12.995713+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18745","created_at":"2026-07-05T09:25:12.995713+00:00"},{"alias_kind":"pith_short_12","alias_value":"3Y7RFGYFG6WP","created_at":"2026-07-05T09:25:12.995713+00:00"},{"alias_kind":"pith_short_16","alias_value":"3Y7RFGYFG6WPQWAS","created_at":"2026-07-05T09:25:12.995713+00:00"},{"alias_kind":"pith_short_8","alias_value":"3Y7RFGYF","created_at":"2026-07-05T09:25:12.995713+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07740","citing_title":"Jet-Long: Efficient Long-Context Extension with Dynamic Bifocal RoPE","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11640","citing_title":"TAROT: Task-Adaptive Refinement of LLM-prior Graphs for Few-shot Tabular Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08476","citing_title":"FlashCP: Load-Balanced Communication-Efficient Context Parallelism for LLM Training","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24298","citing_title":"An Empirical Evaluation of LLM-Generated Code Security Across Prompting Methods","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24300","citing_title":"Enhancing Reliability in LLM-Based Secure Code Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10371","citing_title":"AgentProg: Empowering Long-Horizon GUI Agents with Program-Guided Context Management","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18348","citing_title":"AdaCluster: Adaptive Query-Key Clustering for Sparse Attention in Video Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01381","citing_title":"A framework for analyzing concept representations in neural models","ref_index":165,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G","json":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G.json","graph_json":"https://pith.science/api/pith-number/3Y7RFGYFG6WPQWASZPC3F3ZH2G/graph.json","events_json":"https://pith.science/api/pith-number/3Y7RFGYFG6WPQWASZPC3F3ZH2G/events.json","paper":"https://pith.science/paper/3Y7RFGYF"},"agent_actions":{"view_html":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G","download_json":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G.json","view_paper":"https://pith.science/paper/3Y7RFGYF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18745&json=true","fetch_graph":"https://pith.science/api/pith-number/3Y7RFGYFG6WPQWASZPC3F3ZH2G/graph.json","fetch_events":"https://pith.science/api/pith-number/3Y7RFGYFG6WPQWASZPC3F3ZH2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G/action/storage_attestation","attest_author":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G/action/author_attestation","sign_citation":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G/action/citation_signature","submit_replication":"https://pith.science/pith/3Y7RFGYFG6WPQWASZPC3F3ZH2G/action/replication_record"}},"created_at":"2026-07-05T09:25:12.995713+00:00","updated_at":"2026-07-05T09:25:12.995713+00:00"}