{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:3NJ7KTVB5XWJB34C7VWUPDF7IM","short_pith_number":"pith:3NJ7KTVB","schema_version":"1.0","canonical_sha256":"db53f54ea1edec90ef82fd6d478cbf433082f848bb63e237e067e757ab13fbb4","source":{"kind":"arxiv","id":"2211.04325","version":2},"attestation_state":"computed","paper":{"title":"Will we run out of data? Limits of LLM scaling based on human-generated data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.CY"],"primary_cat":"cs.LG","authors_text":"Anson Ho, Jaime Sevilla, Lennart Heim, Marius Hobbhahn, Pablo Villalobos, Tamay Besiroglu","submitted_at":"2022-10-26T00:28:40Z","abstract_excerpt":"We investigate the potential constraints on LLM scaling posed by the availability of public human-generated text data. We forecast the growing demand for training data based on current trends and estimate the total stock of public human text data. Our findings indicate that if current LLM development trends continue, models will be trained on datasets roughly equal in size to the available stock of public human text data between 2026 and 2032, or slightly earlier if models are overtrained. We explore how progress in language modeling can continue when human-generated text datasets cannot be sc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.04325","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-26T00:28:40Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.CY"],"title_canon_sha256":"fa7e146c6c5cebe18a0c0e861c61a7b35d861ac92cdd14566dedc1d91207cfc4","abstract_canon_sha256":"b165627d427dfdc332b558b4c7f1c9cd63bfd064e8881aabb3c7d67a4d0f947a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:21.806912Z","signature_b64":"wAH+3I5oHJ3Y71zlNcbGxxSUZfkKiG7wvBrVqRBwAbBzdX1N4zBurSSR1XX+lV1DtbglVRoatzRp2R53ORtNBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db53f54ea1edec90ef82fd6d478cbf433082f848bb63e237e067e757ab13fbb4","last_reissued_at":"2026-07-05T08:27:21.806432Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:21.806432Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Will we run out of data? Limits of LLM scaling based on human-generated data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.CY"],"primary_cat":"cs.LG","authors_text":"Anson Ho, Jaime Sevilla, Lennart Heim, Marius Hobbhahn, Pablo Villalobos, Tamay Besiroglu","submitted_at":"2022-10-26T00:28:40Z","abstract_excerpt":"We investigate the potential constraints on LLM scaling posed by the availability of public human-generated text data. We forecast the growing demand for training data based on current trends and estimate the total stock of public human text data. Our findings indicate that if current LLM development trends continue, models will be trained on datasets roughly equal in size to the available stock of public human text data between 2026 and 2032, or slightly earlier if models are overtrained. We explore how progress in language modeling can continue when human-generated text datasets cannot be sc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.04325","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.04325/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.04325","created_at":"2026-07-05T08:27:21.806488+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.04325v2","created_at":"2026-07-05T08:27:21.806488+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.04325","created_at":"2026-07-05T08:27:21.806488+00:00"},{"alias_kind":"pith_short_12","alias_value":"3NJ7KTVB5XWJ","created_at":"2026-07-05T08:27:21.806488+00:00"},{"alias_kind":"pith_short_16","alias_value":"3NJ7KTVB5XWJB34C","created_at":"2026-07-05T08:27:21.806488+00:00"},{"alias_kind":"pith_short_8","alias_value":"3NJ7KTVB","created_at":"2026-07-05T08:27:21.806488+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16246","citing_title":"Demystifying Training-Time Augmentation for Data-Constrained Language Model Pretraining","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10127","citing_title":"Data-Driven Automation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06888","citing_title":"Data-Constrained Language Model Pretraining: Improved Regularization and Scaling Laws","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30774","citing_title":"What Drives Interactive Improvement from Feedback?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12422","citing_title":"Creating and Evaluating K-12 GenAI Assessment Graders Through Context Engineering","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01155","citing_title":"When Data Is Scarce: Scaling Sparse Language Models with Repeated Training","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03920","citing_title":"Enhancing Instructional Quality: Leveraging Computer-Assisted Textual Analysis to Generate In-Depth Insights from Educational Artifacts","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2412.02125","citing_title":"Preference Goal Tuning: Post-Training as Latent Control for Frozen Policies","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04419","citing_title":"Understanding and Mitigating Bias Inheritance in LLM-based Data Augmentation on Downstream Tasks","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2503.08223","citing_title":"Will LLMs Scaling Hit the Wall? Breaking Barriers via Distributed Resources on Massive Edge Devices","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18535","citing_title":"Beyond Scaling: Agents Are Heading to the Edge","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2305.16264","citing_title":"Scaling Data-Constrained Language Models","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16867","citing_title":"The Falcon Series of Open Language Models","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21734","citing_title":"Hierarchical Reasoning Model","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2306.01116","citing_title":"The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data, and Web Data Only","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11733","citing_title":"Position: LLM Inference Should Be Evaluated as Energy-to-Token Production","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23026","citing_title":"AI Hastens Limits to Exponential Growth","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04428","citing_title":"Submodular Ground-Set Pruning: Monotone Tightness and a Non-Monotone Separation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01047","citing_title":"LLM Ghostbusters: Surgical Hallucination Suppression via Adaptive Unlearning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19015","citing_title":"FedProxy: Federated Fine-Tuning of LLMs via Proxy SLMs and Heterogeneity-Aware Fusion","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07474","citing_title":"ForgeVLA: Federated Vision-Language-Action Learning without Language Annotations","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07724","citing_title":"Curated Synthetic Data Doesn't Have to Collapse: A Theoretical Study of Generative Retraining with Pluralistic Preferences","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05135","citing_title":"SenseAI: A Human-in-the-Loop Dataset for RLHF-Aligned Financial Sentiment Reasoning","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM","json":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM.json","graph_json":"https://pith.science/api/pith-number/3NJ7KTVB5XWJB34C7VWUPDF7IM/graph.json","events_json":"https://pith.science/api/pith-number/3NJ7KTVB5XWJB34C7VWUPDF7IM/events.json","paper":"https://pith.science/paper/3NJ7KTVB"},"agent_actions":{"view_html":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM","download_json":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM.json","view_paper":"https://pith.science/paper/3NJ7KTVB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.04325&json=true","fetch_graph":"https://pith.science/api/pith-number/3NJ7KTVB5XWJB34C7VWUPDF7IM/graph.json","fetch_events":"https://pith.science/api/pith-number/3NJ7KTVB5XWJB34C7VWUPDF7IM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM/action/storage_attestation","attest_author":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM/action/author_attestation","sign_citation":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM/action/citation_signature","submit_replication":"https://pith.science/pith/3NJ7KTVB5XWJB34C7VWUPDF7IM/action/replication_record"}},"created_at":"2026-07-05T08:27:21.806488+00:00","updated_at":"2026-07-05T08:27:21.806488+00:00"}