{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JYMI52FYN4IKQSNIVUE6AYSZFX","short_pith_number":"pith:JYMI52FY","schema_version":"1.0","canonical_sha256":"4e188ee8b86f10a849a8ad09e062592dc5ebd714c0fc7a9f363b081f3b03b15f","source":{"kind":"arxiv","id":"2402.07043","version":2},"attestation_state":"computed","paper":{"title":"A Tale of Tails: Model Collapse as a Change of Scaling Laws","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Elvis Dohmatob, Francois Charton, Julia Kempe, Pu Yang, Yunzhen Feng","submitted_at":"2024-02-10T21:06:34Z","abstract_excerpt":"As AI model size grows, neural scaling laws have become a crucial tool to predict the improvements of large models when increasing capacity and the size of original (human or natural) training data. Yet, the widespread use of popular models means that the ecosystem of online data and text will co-evolve to progressively contain increased amounts of synthesized data. In this paper we ask: How will the scaling laws change in the inevitable regime where synthetic data makes its way into the training corpus? Will future models, still improve, or be doomed to degenerate up to total (model) collapse"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07043","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-10T21:06:34Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"75407b03dbc4d745bdf1cb626b1ae529130ceb829c4071f0852640a906940b23","abstract_canon_sha256":"9c2e8498cafe7a163dddaefe1b1dda7b77acbf9bb68e61dbc47ac02c3097d83b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:31.698036Z","signature_b64":"qi9G02S81bb2jvVo9ZQvcaZNzaYBbh7lz7IAwd/2yoOvTHrU74w5G5aWu5LRXdFbE08Iapdo6nM7elwJNd8/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e188ee8b86f10a849a8ad09e062592dc5ebd714c0fc7a9f363b081f3b03b15f","last_reissued_at":"2026-07-05T08:25:31.697489Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:31.697489Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Tale of Tails: Model Collapse as a Change of Scaling Laws","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Elvis Dohmatob, Francois Charton, Julia Kempe, Pu Yang, Yunzhen Feng","submitted_at":"2024-02-10T21:06:34Z","abstract_excerpt":"As AI model size grows, neural scaling laws have become a crucial tool to predict the improvements of large models when increasing capacity and the size of original (human or natural) training data. Yet, the widespread use of popular models means that the ecosystem of online data and text will co-evolve to progressively contain increased amounts of synthesized data. In this paper we ask: How will the scaling laws change in the inevitable regime where synthetic data makes its way into the training corpus? Will future models, still improve, or be doomed to degenerate up to total (model) collapse"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07043","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07043","created_at":"2026-07-05T08:25:31.697576+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07043v2","created_at":"2026-07-05T08:25:31.697576+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07043","created_at":"2026-07-05T08:25:31.697576+00:00"},{"alias_kind":"pith_short_12","alias_value":"JYMI52FYN4IK","created_at":"2026-07-05T08:25:31.697576+00:00"},{"alias_kind":"pith_short_16","alias_value":"JYMI52FYN4IKQSNI","created_at":"2026-07-05T08:25:31.697576+00:00"},{"alias_kind":"pith_short_8","alias_value":"JYMI52FY","created_at":"2026-07-05T08:25:31.697576+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28343","citing_title":"The Crowded Embedding Space: A Mean-Field Mechanism for Emergent Marginalization in Retrieval-Augmented Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28438","citing_title":"When AI Reviews Its Own Code: Recursive Self-Training Collapse in Code LLMs","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2406.20094","citing_title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21853","citing_title":"Generative artificial intelligence reduces social welfare through model collapse","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX","json":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX.json","graph_json":"https://pith.science/api/pith-number/JYMI52FYN4IKQSNIVUE6AYSZFX/graph.json","events_json":"https://pith.science/api/pith-number/JYMI52FYN4IKQSNIVUE6AYSZFX/events.json","paper":"https://pith.science/paper/JYMI52FY"},"agent_actions":{"view_html":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX","download_json":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX.json","view_paper":"https://pith.science/paper/JYMI52FY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07043&json=true","fetch_graph":"https://pith.science/api/pith-number/JYMI52FYN4IKQSNIVUE6AYSZFX/graph.json","fetch_events":"https://pith.science/api/pith-number/JYMI52FYN4IKQSNIVUE6AYSZFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX/action/storage_attestation","attest_author":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX/action/author_attestation","sign_citation":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX/action/citation_signature","submit_replication":"https://pith.science/pith/JYMI52FYN4IKQSNIVUE6AYSZFX/action/replication_record"}},"created_at":"2026-07-05T08:25:31.697576+00:00","updated_at":"2026-07-05T08:25:31.697576+00:00"}