{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7YQLDZT62QJ3LT2BHK7LHCP4XR","short_pith_number":"pith:7YQLDZT6","schema_version":"1.0","canonical_sha256":"fe20b1e67ed413b5cf413abeb389fcbc4e07ad290decf890f697e554b61a4685","source":{"kind":"arxiv","id":"2308.16137","version":7},"attestation_state":"computed","paper":{"title":"LM-Infinite: Zero-Shot Extreme Length Generalization for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chi Han, Hao Peng, Heng Ji, Qifan Wang, Sinong Wang, Wenhan Xiong, Yu Chen","submitted_at":"2023-08-30T16:47:51Z","abstract_excerpt":"Today's large language models (LLMs) typically train on short text segments (e.g., <4K tokens) due to the quadratic complexity of their Transformer architectures. As a result, their performance suffers drastically on inputs longer than those encountered during training, substantially limiting their applications in real-world tasks involving long contexts such as encoding scientific articles, code repositories, or long dialogues. Through theoretical analysis and empirical investigation, this work identifies three major factors contributing to this length generalization failure. Our theoretical "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.16137","kind":"arxiv","version":7},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-30T16:47:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3ff6c1bca29a917748467efb2979c8a8c4c5940f94b7fd31028934d5e9a59c39","abstract_canon_sha256":"da39e613e7b58abf2def73a607d4418fd284a25d064df7252f5a8127bc52342e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:19.353903Z","signature_b64":"O5RhzDTF8Q1kiQeU2rlNdP2ZucoCtnd3F+Rh3r7pRYa+X595+MBgot4ssjdZ++MrTDr3rqQyktKyItgMB6YtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fe20b1e67ed413b5cf413abeb389fcbc4e07ad290decf890f697e554b61a4685","last_reissued_at":"2026-07-05T08:36:19.353471Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:19.353471Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LM-Infinite: Zero-Shot Extreme Length Generalization for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chi Han, Hao Peng, Heng Ji, Qifan Wang, Sinong Wang, Wenhan Xiong, Yu Chen","submitted_at":"2023-08-30T16:47:51Z","abstract_excerpt":"Today's large language models (LLMs) typically train on short text segments (e.g., <4K tokens) due to the quadratic complexity of their Transformer architectures. As a result, their performance suffers drastically on inputs longer than those encountered during training, substantially limiting their applications in real-world tasks involving long contexts such as encoding scientific articles, code repositories, or long dialogues. Through theoretical analysis and empirical investigation, this work identifies three major factors contributing to this length generalization failure. Our theoretical "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.16137","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.16137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.16137","created_at":"2026-07-05T08:36:19.353531+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.16137v7","created_at":"2026-07-05T08:36:19.353531+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.16137","created_at":"2026-07-05T08:36:19.353531+00:00"},{"alias_kind":"pith_short_12","alias_value":"7YQLDZT62QJ3","created_at":"2026-07-05T08:36:19.353531+00:00"},{"alias_kind":"pith_short_16","alias_value":"7YQLDZT62QJ3LT2B","created_at":"2026-07-05T08:36:19.353531+00:00"},{"alias_kind":"pith_short_8","alias_value":"7YQLDZT6","created_at":"2026-07-05T08:36:19.353531+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24734","citing_title":"Task Decomposition for Efficient Annotation","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07878","citing_title":"Still: Amortized KV Cache Compaction in a Single Forward Pass","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01101","citing_title":"Soft-NBCE: Entropy-Weighted Chunk Fusion for Long-Context","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24279","citing_title":"ContextEcho: A Benchmark for Persona Drift in Long Agentic-Coding Sessions","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29563","citing_title":"Coverage-Driven KV Cache Eviction for Efficient and Improved Inference of LLM","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":186,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17247","citing_title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13753","citing_title":"LongRoPE: Extending LLM Context Window Beyond 2 Million Tokens","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02069","citing_title":"PyramidKV: Dynamic KV Cache Compression based on Pyramidal Information Funneling","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08840","citing_title":"ReST-KV: Robust KV Cache Eviction with Layer-wise Output Reconstruction and Spatial-Temporal Smoothing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2309.00071","citing_title":"YaRN: Efficient Context Window Extension of Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09649","citing_title":"Make Each Token Count: Towards Improving Long-Context Performance with KV Cache Eviction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06611","citing_title":"The Structural Origin of Attention Sink: Variance Discrepancy, Super Neurons, and Dimension Disparity","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02262","citing_title":"WindowQuant: Mixed-Precision KV Cache Quantization based on Window-Level Similarity for VLMs Inference Optimization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2404.06654","citing_title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR","json":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR.json","graph_json":"https://pith.science/api/pith-number/7YQLDZT62QJ3LT2BHK7LHCP4XR/graph.json","events_json":"https://pith.science/api/pith-number/7YQLDZT62QJ3LT2BHK7LHCP4XR/events.json","paper":"https://pith.science/paper/7YQLDZT6"},"agent_actions":{"view_html":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR","download_json":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR.json","view_paper":"https://pith.science/paper/7YQLDZT6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.16137&json=true","fetch_graph":"https://pith.science/api/pith-number/7YQLDZT62QJ3LT2BHK7LHCP4XR/graph.json","fetch_events":"https://pith.science/api/pith-number/7YQLDZT62QJ3LT2BHK7LHCP4XR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR/action/storage_attestation","attest_author":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR/action/author_attestation","sign_citation":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR/action/citation_signature","submit_replication":"https://pith.science/pith/7YQLDZT62QJ3LT2BHK7LHCP4XR/action/replication_record"}},"created_at":"2026-07-05T08:36:19.353531+00:00","updated_at":"2026-07-05T08:36:19.353531+00:00"}