{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5OKDIZC77Y522UIFFXAIKO5ZO4","short_pith_number":"pith:5OKDIZC7","schema_version":"1.0","canonical_sha256":"eb9434645ffe3bad51052dc0853bb9771e2de1f312dd4ccbe78d69a2fb1fa027","source":{"kind":"arxiv","id":"2402.17463","version":2},"attestation_state":"computed","paper":{"title":"Training-Free Long-Context Scaling of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chang Zhou, Chenxin An, Fei Huang, Jun Zhang, Lingpeng Kong, Shansan Gong, Xipeng Qiu","submitted_at":"2024-02-27T12:39:23Z","abstract_excerpt":"The ability of Large Language Models (LLMs) to process and generate coherent text is markedly weakened when the number of input tokens exceeds their pretraining length. Given the expensive overhead of finetuning large-scale models with longer sequences, we propose Dual Chunk Attention (DCA), which enables Llama2 70B to support context windows of more than 100k tokens without continual training. By decomposing the attention computation for long sequences into chunk-based modules, DCA manages to effectively capture the relative positional information of tokens within the same chunk (Intra-Chunk)"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17463","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-27T12:39:23Z","cross_cats_sorted":[],"title_canon_sha256":"47ab3b0acef92d0d08aac7babe79ccb6a7d7d340f3e7f42cf6108a5b71d53f17","abstract_canon_sha256":"7f96296e894849bbdb0259477b05ba9a92fd5099d853f476103d95f5a71cdbf7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:24.183904Z","signature_b64":"KmToU6TbczwzpvNKv22pOS7GbQJCaEQujC6ImQF02JDiRQ/yZYPUzMobl8gVgJcl8LGXtU7dw5Q0qhghhreNBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb9434645ffe3bad51052dc0853bb9771e2de1f312dd4ccbe78d69a2fb1fa027","last_reissued_at":"2026-07-05T08:24:24.183399Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:24.183399Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training-Free Long-Context Scaling of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chang Zhou, Chenxin An, Fei Huang, Jun Zhang, Lingpeng Kong, Shansan Gong, Xipeng Qiu","submitted_at":"2024-02-27T12:39:23Z","abstract_excerpt":"The ability of Large Language Models (LLMs) to process and generate coherent text is markedly weakened when the number of input tokens exceeds their pretraining length. Given the expensive overhead of finetuning large-scale models with longer sequences, we propose Dual Chunk Attention (DCA), which enables Llama2 70B to support context windows of more than 100k tokens without continual training. By decomposing the attention computation for long sequences into chunk-based modules, DCA manages to effectively capture the relative positional information of tokens within the same chunk (Intra-Chunk)"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17463","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17463/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17463","created_at":"2026-07-05T08:24:24.183465+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17463v2","created_at":"2026-07-05T08:24:24.183465+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17463","created_at":"2026-07-05T08:24:24.183465+00:00"},{"alias_kind":"pith_short_12","alias_value":"5OKDIZC77Y52","created_at":"2026-07-05T08:24:24.183465+00:00"},{"alias_kind":"pith_short_16","alias_value":"5OKDIZC77Y522UIF","created_at":"2026-07-05T08:24:24.183465+00:00"},{"alias_kind":"pith_short_8","alias_value":"5OKDIZC7","created_at":"2026-07-05T08:24:24.183465+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07740","citing_title":"Jet-Long: Efficient Long-Context Extension with Dynamic Bifocal RoPE","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2412.15115","citing_title":"Qwen2.5 Technical Report","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.13047","citing_title":"Multi-Model Synthetic Training for Mission-Critical Small Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13933","citing_title":"HyMem: Hybrid Memory Architecture with Dynamic Retrieval Scheduling","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02259","citing_title":"MemAgent: Reshaping Long-Context LLM with Multi-Conv RL-based Memory Agent","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2501.15383","citing_title":"Qwen2.5-1M Technical Report","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04595","citing_title":"A Queueing-Theoretic Framework for Stability Analysis of LLM Inference with KV Cache Memory Constraints","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17366","citing_title":"ArgBench: Benchmarking LLMs on Computational Argumentation Tasks","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09388","citing_title":"Qwen3 Technical Report","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4","json":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4.json","graph_json":"https://pith.science/api/pith-number/5OKDIZC77Y522UIFFXAIKO5ZO4/graph.json","events_json":"https://pith.science/api/pith-number/5OKDIZC77Y522UIFFXAIKO5ZO4/events.json","paper":"https://pith.science/paper/5OKDIZC7"},"agent_actions":{"view_html":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4","download_json":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4.json","view_paper":"https://pith.science/paper/5OKDIZC7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17463&json=true","fetch_graph":"https://pith.science/api/pith-number/5OKDIZC77Y522UIFFXAIKO5ZO4/graph.json","fetch_events":"https://pith.science/api/pith-number/5OKDIZC77Y522UIFFXAIKO5ZO4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4/action/storage_attestation","attest_author":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4/action/author_attestation","sign_citation":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4/action/citation_signature","submit_replication":"https://pith.science/pith/5OKDIZC77Y522UIFFXAIKO5ZO4/action/replication_record"}},"created_at":"2026-07-05T08:24:24.183465+00:00","updated_at":"2026-07-05T08:24:24.183465+00:00"}