{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AKQW2IBSZYK2D6ZFKR5ONXESHN","short_pith_number":"pith:AKQW2IBS","schema_version":"1.0","canonical_sha256":"02a16d2032ce15a1fb25547ae6dc923b478ba0980eaa21ab4c6eb334d6913c0f","source":{"kind":"arxiv","id":"2309.16039","version":3},"attestation_state":"computed","paper":{"title":"Effective Long-Context Scaling of Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Angela Fan, Barlas Oguz, Han Fang, Hao Ma, Hejia Zhang, Igor Molybog, Jingyu Liu, Karthik Abinav Sankararaman, Kshitiz Malik, Louis Martin, Madian Khabsa, Mike Lewis, Prajjwal Bhargava, Rashi Rungta, Rui Hou, Sergey Edunov, Sharan Narang, Shruti Bhosale, Sinong Wang, Wenhan Xiong, Yashar Mehdad","submitted_at":"2023-09-27T21:41:49Z","abstract_excerpt":"We present a series of long-context LLMs that support effective context windows of up to 32,768 tokens. Our model series are built through continual pretraining from Llama 2 with longer training sequences and on a dataset where long texts are upsampled. We perform extensive evaluation on language modeling, synthetic context probing tasks, and a wide range of research benchmarks. On research benchmarks, our models achieve consistent improvements on most regular tasks and significant improvements on long-context tasks over Llama 2. Notably, with a cost-effective instruction tuning procedure that"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16039","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-27T21:41:49Z","cross_cats_sorted":[],"title_canon_sha256":"8eb7dc28f1d7f8c2ff7f6d475e8bb51ccfc3030a6d2ba0a3c9e4605f0d343f6f","abstract_canon_sha256":"7d41460099cf26210a5f486dd5c6ed5605d52f709dcf2720bccf46121f818862"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:18.720714Z","signature_b64":"LudLmAolsEUtkLoA80F189Fs33ZymTxaiQGPMp5gP6Ypq3Xt5MNCsFeh7rFvwrT0gKEuMATLKnXTkBEfgrYdCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"02a16d2032ce15a1fb25547ae6dc923b478ba0980eaa21ab4c6eb334d6913c0f","last_reissued_at":"2026-07-05T07:12:18.719956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:18.719956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Effective Long-Context Scaling of Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Angela Fan, Barlas Oguz, Han Fang, Hao Ma, Hejia Zhang, Igor Molybog, Jingyu Liu, Karthik Abinav Sankararaman, Kshitiz Malik, Louis Martin, Madian Khabsa, Mike Lewis, Prajjwal Bhargava, Rashi Rungta, Rui Hou, Sergey Edunov, Sharan Narang, Shruti Bhosale, Sinong Wang, Wenhan Xiong, Yashar Mehdad","submitted_at":"2023-09-27T21:41:49Z","abstract_excerpt":"We present a series of long-context LLMs that support effective context windows of up to 32,768 tokens. Our model series are built through continual pretraining from Llama 2 with longer training sequences and on a dataset where long texts are upsampled. We perform extensive evaluation on language modeling, synthetic context probing tasks, and a wide range of research benchmarks. On research benchmarks, our models achieve consistent improvements on most regular tasks and significant improvements on long-context tasks over Llama 2. Notably, with a cost-effective instruction tuning procedure that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16039","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16039/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16039","created_at":"2026-07-05T07:12:18.720033+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16039v3","created_at":"2026-07-05T07:12:18.720033+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16039","created_at":"2026-07-05T07:12:18.720033+00:00"},{"alias_kind":"pith_short_12","alias_value":"AKQW2IBSZYK2","created_at":"2026-07-05T07:12:18.720033+00:00"},{"alias_kind":"pith_short_16","alias_value":"AKQW2IBSZYK2D6ZF","created_at":"2026-07-05T07:12:18.720033+00:00"},{"alias_kind":"pith_short_8","alias_value":"AKQW2IBS","created_at":"2026-07-05T07:12:18.720033+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05345","citing_title":"PJ-RoPE: A Fourier-Jet-Affine Position Space for Relative Attention","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30460","citing_title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30460","citing_title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2403.20208","citing_title":"Unlock the Potential of Large Language Models for Predictive Tabular Tasks in Data Science with Table-Specific Pretraining","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15115","citing_title":"Qwen2.5 Technical Report","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07143","citing_title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12635","citing_title":"Positional Encoding via Token-Aware Phase Attention","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2406.07887","citing_title":"An Empirical Study of Mamba-based Language Models","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":205,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07805","citing_title":"Group Representational Position Encoding","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13189","citing_title":"MoBA: Mixture of Block Attention for Long-Context LLMs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2403.17297","citing_title":"InternLM2 Technical Report","ref_index":151,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02259","citing_title":"MemAgent: Reshaping Long-Context LLM with Multi-Conv RL-based Memory Agent","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2501.15383","citing_title":"Qwen2.5-1M Technical Report","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23885","citing_title":"Towards Real-World Document Parsing via Realistic Scene Synthesis and Document-Aware Training","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26692","citing_title":"Kimi Linear: An Expressive, Efficient Attention Architecture","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2404.06395","citing_title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2403.04652","citing_title":"Yi: Open Foundation Models by 01.AI","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2406.12793","citing_title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07809","citing_title":"PolicyLong: Towards On-Policy Context Extension","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2408.06072","citing_title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16864","citing_title":"HieraSparse: Hierarchical Semi-Structured Sparse KV Attention","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09388","citing_title":"Qwen3 Technical Report","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN","json":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN.json","graph_json":"https://pith.science/api/pith-number/AKQW2IBSZYK2D6ZFKR5ONXESHN/graph.json","events_json":"https://pith.science/api/pith-number/AKQW2IBSZYK2D6ZFKR5ONXESHN/events.json","paper":"https://pith.science/paper/AKQW2IBS"},"agent_actions":{"view_html":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN","download_json":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN.json","view_paper":"https://pith.science/paper/AKQW2IBS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16039&json=true","fetch_graph":"https://pith.science/api/pith-number/AKQW2IBSZYK2D6ZFKR5ONXESHN/graph.json","fetch_events":"https://pith.science/api/pith-number/AKQW2IBSZYK2D6ZFKR5ONXESHN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN/action/storage_attestation","attest_author":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN/action/author_attestation","sign_citation":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN/action/citation_signature","submit_replication":"https://pith.science/pith/AKQW2IBSZYK2D6ZFKR5ONXESHN/action/replication_record"}},"created_at":"2026-07-05T07:12:18.720033+00:00","updated_at":"2026-07-05T07:12:18.720033+00:00"}