{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WYRFALICQ3XV6ME6KMU4WMKUFU","short_pith_number":"pith:WYRFALIC","schema_version":"1.0","canonical_sha256":"b622502d0286ef5f309e5329cb31542d30d954dfe82f8d60c369c041779af271","source":{"kind":"arxiv","id":"2310.10638","version":6},"attestation_state":"computed","paper":{"title":"In-context Pretraining: Language Modeling Beyond Document Boundaries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Gergely Szilvasy, Luke Zettlemoyer, Margaret Li, Maria Lomeli, Mike Lewis, Noah A. Smith, Rich James, Scott Yih, Sewon Min, Weijia Shi, Xi Victoria Lin","submitted_at":"2023-10-16T17:57:12Z","abstract_excerpt":"Large language models (LMs) are currently trained to predict tokens given document prefixes, enabling them to directly perform long-form generation and prompting-style tasks which can be reduced to document completion. Existing pretraining pipelines train LMs by concatenating random sets of short documents to create input contexts but the prior documents provide no signal for predicting the next document. We instead present In-Context Pretraining, a new approach where language models are pretrained on a sequence of related documents, thereby explicitly encouraging them to read and reason acros"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10638","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T17:57:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e60e7ad6d0b94385b0b388cae46b924a7bdede6919063002ec5335cab65f79c7","abstract_canon_sha256":"0c37c294fbd04b8898a60467a00cf31817e7c3bd94923b1eb751222355c90ea0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:38.414240Z","signature_b64":"bgKwkflcEIukn3GuTmO5GbQwHRiiRaw6WArJ6hjHPqbsSKOdB6i8sVhjchYLzb0xhjOPa2JinYlr3jeFJHFPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b622502d0286ef5f309e5329cb31542d30d954dfe82f8d60c369c041779af271","last_reissued_at":"2026-07-05T08:35:38.413754Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:38.413754Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"In-context Pretraining: Language Modeling Beyond Document Boundaries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Gergely Szilvasy, Luke Zettlemoyer, Margaret Li, Maria Lomeli, Mike Lewis, Noah A. Smith, Rich James, Scott Yih, Sewon Min, Weijia Shi, Xi Victoria Lin","submitted_at":"2023-10-16T17:57:12Z","abstract_excerpt":"Large language models (LMs) are currently trained to predict tokens given document prefixes, enabling them to directly perform long-form generation and prompting-style tasks which can be reduced to document completion. Existing pretraining pipelines train LMs by concatenating random sets of short documents to create input contexts but the prior documents provide no signal for predicting the next document. We instead present In-Context Pretraining, a new approach where language models are pretrained on a sequence of related documents, thereby explicitly encouraging them to read and reason acros"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10638","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10638","created_at":"2026-07-05T08:35:38.413812+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10638v6","created_at":"2026-07-05T08:35:38.413812+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10638","created_at":"2026-07-05T08:35:38.413812+00:00"},{"alias_kind":"pith_short_12","alias_value":"WYRFALICQ3XV","created_at":"2026-07-05T08:35:38.413812+00:00"},{"alias_kind":"pith_short_16","alias_value":"WYRFALICQ3XV6ME6","created_at":"2026-07-05T08:35:38.413812+00:00"},{"alias_kind":"pith_short_8","alias_value":"WYRFALIC","created_at":"2026-07-05T08:35:38.413812+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30460","citing_title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30460","citing_title":"HSAP: A Hierarchical Sequence-aware Parallelism for Hybrid-Context Generative Models","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2401.11817","citing_title":"Hallucination is Inevitable: An Innate Limitation of Large Language Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2311.05232","citing_title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions","ref_index":289,"is_internal_anchor":false},{"citing_arxiv_id":"2401.08281","citing_title":"The Faiss library","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU","json":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU.json","graph_json":"https://pith.science/api/pith-number/WYRFALICQ3XV6ME6KMU4WMKUFU/graph.json","events_json":"https://pith.science/api/pith-number/WYRFALICQ3XV6ME6KMU4WMKUFU/events.json","paper":"https://pith.science/paper/WYRFALIC"},"agent_actions":{"view_html":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU","download_json":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU.json","view_paper":"https://pith.science/paper/WYRFALIC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10638&json=true","fetch_graph":"https://pith.science/api/pith-number/WYRFALICQ3XV6ME6KMU4WMKUFU/graph.json","fetch_events":"https://pith.science/api/pith-number/WYRFALICQ3XV6ME6KMU4WMKUFU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU/action/storage_attestation","attest_author":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU/action/author_attestation","sign_citation":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU/action/citation_signature","submit_replication":"https://pith.science/pith/WYRFALICQ3XV6ME6KMU4WMKUFU/action/replication_record"}},"created_at":"2026-07-05T08:35:38.413812+00:00","updated_at":"2026-07-05T08:35:38.413812+00:00"}