{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H4HLAQ53ZJIXYUQLVMJIWJQXV7","short_pith_number":"pith:H4HLAQ53","schema_version":"1.0","canonical_sha256":"3f0eb043bbca517c520bab128b2617afcd4b3d9d56cb7f60fa87811e063e395e","source":{"kind":"arxiv","id":"2401.01325","version":3},"attestation_state":"computed","paper":{"title":"LLM Maybe LongLM: Self-Extend LLM Context Window Without Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chia-Yuan Chang, Hongye Jin, Huiyuan Chen, Jingfeng Yang, Xia Hu, Xiaotian Han, Zhimeng Jiang, Zirui Liu","submitted_at":"2024-01-02T18:30:51Z","abstract_excerpt":"It is well known that LLMs cannot generalize well to long contexts whose lengths are larger than the training sequence length. This poses challenges when employing LLMs for processing long input sequences during inference. In this work, we argue that LLMs themselves have inherent capabilities to handle long contexts without fine-tuning. To achieve this goal, we propose SelfExtend to extend the context window of LLMs by constructing bi-level attention information: the grouped attention and the neighbor attention. The grouped attention captures the dependencies among tokens that are far apart, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.01325","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-02T18:30:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"255b2929515cdcb9f1d8f00be7ef5a53361a0b0dc980d2ed4dd71895c390fb0f","abstract_canon_sha256":"1830ee6af99f3705ef5016edbcd1a2ee00aaaa26fe93bb32f133197ba31db732"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:31.943162Z","signature_b64":"+XdLC45CHH9+88LTYIY4e5WOL6d0ajd1VhBXSFRDl3Ue+eZSRqhCWhIDe2Wze1xX/Vc2t5LPuXZnpNphRJXjCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f0eb043bbca517c520bab128b2617afcd4b3d9d56cb7f60fa87811e063e395e","last_reissued_at":"2026-07-05T08:42:31.942748Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:31.942748Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Maybe LongLM: Self-Extend LLM Context Window Without Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chia-Yuan Chang, Hongye Jin, Huiyuan Chen, Jingfeng Yang, Xia Hu, Xiaotian Han, Zhimeng Jiang, Zirui Liu","submitted_at":"2024-01-02T18:30:51Z","abstract_excerpt":"It is well known that LLMs cannot generalize well to long contexts whose lengths are larger than the training sequence length. This poses challenges when employing LLMs for processing long input sequences during inference. In this work, we argue that LLMs themselves have inherent capabilities to handle long contexts without fine-tuning. To achieve this goal, we propose SelfExtend to extend the context window of LLMs by constructing bi-level attention information: the grouped attention and the neighbor attention. The grouped attention captures the dependencies among tokens that are far apart, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.01325","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.01325/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.01325","created_at":"2026-07-05T08:42:31.942802+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.01325v3","created_at":"2026-07-05T08:42:31.942802+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.01325","created_at":"2026-07-05T08:42:31.942802+00:00"},{"alias_kind":"pith_short_12","alias_value":"H4HLAQ53ZJIX","created_at":"2026-07-05T08:42:31.942802+00:00"},{"alias_kind":"pith_short_16","alias_value":"H4HLAQ53ZJIXYUQL","created_at":"2026-07-05T08:42:31.942802+00:00"},{"alias_kind":"pith_short_8","alias_value":"H4HLAQ53","created_at":"2026-07-05T08:42:31.942802+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07740","citing_title":"Jet-Long: Efficient Long-Context Extension with Dynamic Bifocal RoPE","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05748","citing_title":"UNIVID: Unified Vision-Language Model for Video Moderation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28154","citing_title":"Robo-Blocks: Generative Scaffolding in End-User Design and Programming of Social Robots","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11295","citing_title":"The Prompt Engineering Report Distilled: Quick Start Guide for Life Sciences","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11295","citing_title":"The Prompt Engineering Report Distilled: Quick Start Guide for Life Sciences","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14487","citing_title":"Head Forcing: Long Autoregressive Video Generation via Head Heterogeneity","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13831","citing_title":"Training Long-Context Vision-Language Models Effectively with Generalization Beyond 128K Context","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14469","citing_title":"SnapKV: LLM Knows What You are Looking for Before Generation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10544","citing_title":"Where Does Long-Context Supervision Actually Go? Effective-Context Exposure Balancing","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24608","citing_title":"Learning to Route Queries to Heads for Attention-based Re-ranking with Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06351","citing_title":"SIGMA-ASL: Sensor-Integrated Multimodal Dataset for Sign Language Recognition","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05066","citing_title":"The Impossibility Triangle of Long-Context Modeling","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7","json":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7.json","graph_json":"https://pith.science/api/pith-number/H4HLAQ53ZJIXYUQLVMJIWJQXV7/graph.json","events_json":"https://pith.science/api/pith-number/H4HLAQ53ZJIXYUQLVMJIWJQXV7/events.json","paper":"https://pith.science/paper/H4HLAQ53"},"agent_actions":{"view_html":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7","download_json":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7.json","view_paper":"https://pith.science/paper/H4HLAQ53","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.01325&json=true","fetch_graph":"https://pith.science/api/pith-number/H4HLAQ53ZJIXYUQLVMJIWJQXV7/graph.json","fetch_events":"https://pith.science/api/pith-number/H4HLAQ53ZJIXYUQLVMJIWJQXV7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7/action/storage_attestation","attest_author":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7/action/author_attestation","sign_citation":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7/action/citation_signature","submit_replication":"https://pith.science/pith/H4HLAQ53ZJIXYUQLVMJIWJQXV7/action/replication_record"}},"created_at":"2026-07-05T08:42:31.942802+00:00","updated_at":"2026-07-05T08:42:31.942802+00:00"}