{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PKAC2ZU6RFPVQPQWKZXNYJSLH6","short_pith_number":"pith:PKAC2ZU6","schema_version":"1.0","canonical_sha256":"7a802d669e895f583e16566edc264b3fa0a9adc8755f739387ea5f36f1628f53","source":{"kind":"arxiv","id":"2412.17646","version":1},"attestation_state":"computed","paper":{"title":"Rate of Model Collapse in Recursive Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aditya Nanda Kishore Khandavally, Ananda Theertha Suresh, Andrew Thangaraj","submitted_at":"2024-12-23T15:21:50Z","abstract_excerpt":"Given the ease of creating synthetic data from machine learning models, new models can be potentially trained on synthetic data generated by previous models. This recursive training process raises concerns about the long-term impact on model quality. As models are recursively trained on generated data from previous rounds, their ability to capture the nuances of the original human-generated data may degrade. This is often referred to as \\emph{model collapse}. In this work, we ask how fast model collapse occurs for some well-studied distribution families under maximum likelihood (ML or near ML)"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17646","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-23T15:21:50Z","cross_cats_sorted":["cs.IT","math.IT","stat.ML"],"title_canon_sha256":"0da3b6a43d9ce50ab3319e275ebb9c2f21e3ed7eda9f49f9422815ff901e9c58","abstract_canon_sha256":"d5576105e7b98de3c1517402f72b85b8a8f5cca2ba3ee489107735f61e89f950"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:25.438483Z","signature_b64":"Bappufdbf0dMwJ9erzKSEaVVtopp+PaIaxdQ2utUqF9Ub+kgeAbg3RpDT1svpxeRBp/8fUfFbBHxGA7bOko9DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a802d669e895f583e16566edc264b3fa0a9adc8755f739387ea5f36f1628f53","last_reissued_at":"2026-07-05T09:53:25.438059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:25.438059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rate of Model Collapse in Recursive Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aditya Nanda Kishore Khandavally, Ananda Theertha Suresh, Andrew Thangaraj","submitted_at":"2024-12-23T15:21:50Z","abstract_excerpt":"Given the ease of creating synthetic data from machine learning models, new models can be potentially trained on synthetic data generated by previous models. This recursive training process raises concerns about the long-term impact on model quality. As models are recursively trained on generated data from previous rounds, their ability to capture the nuances of the original human-generated data may degrade. This is often referred to as \\emph{model collapse}. In this work, we ask how fast model collapse occurs for some well-studied distribution families under maximum likelihood (ML or near ML)"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17646","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17646/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17646","created_at":"2026-07-05T09:53:25.438115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17646v1","created_at":"2026-07-05T09:53:25.438115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17646","created_at":"2026-07-05T09:53:25.438115+00:00"},{"alias_kind":"pith_short_12","alias_value":"PKAC2ZU6RFPV","created_at":"2026-07-05T09:53:25.438115+00:00"},{"alias_kind":"pith_short_16","alias_value":"PKAC2ZU6RFPVQPQW","created_at":"2026-07-05T09:53:25.438115+00:00"},{"alias_kind":"pith_short_8","alias_value":"PKAC2ZU6","created_at":"2026-07-05T09:53:25.438115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28438","citing_title":"When AI Reviews Its Own Code: Recursive Self-Training Collapse in Code LLMs","ref_index":153,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6","json":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6.json","graph_json":"https://pith.science/api/pith-number/PKAC2ZU6RFPVQPQWKZXNYJSLH6/graph.json","events_json":"https://pith.science/api/pith-number/PKAC2ZU6RFPVQPQWKZXNYJSLH6/events.json","paper":"https://pith.science/paper/PKAC2ZU6"},"agent_actions":{"view_html":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6","download_json":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6.json","view_paper":"https://pith.science/paper/PKAC2ZU6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17646&json=true","fetch_graph":"https://pith.science/api/pith-number/PKAC2ZU6RFPVQPQWKZXNYJSLH6/graph.json","fetch_events":"https://pith.science/api/pith-number/PKAC2ZU6RFPVQPQWKZXNYJSLH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6/action/storage_attestation","attest_author":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6/action/author_attestation","sign_citation":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6/action/citation_signature","submit_replication":"https://pith.science/pith/PKAC2ZU6RFPVQPQWKZXNYJSLH6/action/replication_record"}},"created_at":"2026-07-05T09:53:25.438115+00:00","updated_at":"2026-07-05T09:53:25.438115+00:00"}