{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GRTXIRD7WJPKAJERWRTCJ4AMQC","short_pith_number":"pith:GRTXIRD7","schema_version":"1.0","canonical_sha256":"346774447fb25ea02491b46624f00c80b197b1a3dab7b838d2820d44589f21a3","source":{"kind":"arxiv","id":"2408.16333","version":1},"attestation_state":"computed","paper":{"title":"Self-Improving Diffusion Models with Synthetic Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ahmed Imtiaz Humayun, John Collomosse, Richard Baraniuk, Shruti Agarwal, Sina Alemohammad","submitted_at":"2024-08-29T08:12:18Z","abstract_excerpt":"The artificial intelligence (AI) world is running out of real data for training increasingly large generative models, resulting in accelerating pressure to train on synthetic data. Unfortunately, training new generative models with synthetic data from current or past generation models creates an autophagous (self-consuming) loop that degrades the quality and/or diversity of the synthetic data in what has been termed model autophagy disorder (MAD) and model collapse. Current thinking around model autophagy recommends that synthetic data is to be avoided for model training lest the system deteri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.16333","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-29T08:12:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d83a3327608635f887afd5c5e00eb55a0f93c1dfe697a665730e1b46b1ec7adc","abstract_canon_sha256":"db20d654b0f98b0be5da06c86592a12d07a6fb39a6d719807b99ce503077a64c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:39.586118Z","signature_b64":"P3qPGxRRqkMe4rq4o6Uim2Aez1Ml8IStQpCAJY0DBNc2UeNEdfzfjyUlWaDAvW9OhbjbUgyyUwKWOrIt/rgLAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"346774447fb25ea02491b46624f00c80b197b1a3dab7b838d2820d44589f21a3","last_reissued_at":"2026-07-05T09:00:39.585632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:39.585632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Improving Diffusion Models with Synthetic Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ahmed Imtiaz Humayun, John Collomosse, Richard Baraniuk, Shruti Agarwal, Sina Alemohammad","submitted_at":"2024-08-29T08:12:18Z","abstract_excerpt":"The artificial intelligence (AI) world is running out of real data for training increasingly large generative models, resulting in accelerating pressure to train on synthetic data. Unfortunately, training new generative models with synthetic data from current or past generation models creates an autophagous (self-consuming) loop that degrades the quality and/or diversity of the synthetic data in what has been termed model autophagy disorder (MAD) and model collapse. Current thinking around model autophagy recommends that synthetic data is to be avoided for model training lest the system deteri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.16333","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.16333/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.16333","created_at":"2026-07-05T09:00:39.585690+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.16333v1","created_at":"2026-07-05T09:00:39.585690+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.16333","created_at":"2026-07-05T09:00:39.585690+00:00"},{"alias_kind":"pith_short_12","alias_value":"GRTXIRD7WJPK","created_at":"2026-07-05T09:00:39.585690+00:00"},{"alias_kind":"pith_short_16","alias_value":"GRTXIRD7WJPKAJER","created_at":"2026-07-05T09:00:39.585690+00:00"},{"alias_kind":"pith_short_8","alias_value":"GRTXIRD7","created_at":"2026-07-05T09:00:39.585690+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08802","citing_title":"Active Flow Expansion for Out-of-Distribution Discovery: from Theory to Molecules","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06020","citing_title":"ReSAGE-PAR: Representational Similarity Assessment for Generative Expansion in Pedestrian Attribute Recognition","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06501","citing_title":"Enhancing Malware Detection with Generative AI: Using Variational Autoencoders to Boost Machine Learning Classifiers' Performance","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC","json":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC.json","graph_json":"https://pith.science/api/pith-number/GRTXIRD7WJPKAJERWRTCJ4AMQC/graph.json","events_json":"https://pith.science/api/pith-number/GRTXIRD7WJPKAJERWRTCJ4AMQC/events.json","paper":"https://pith.science/paper/GRTXIRD7"},"agent_actions":{"view_html":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC","download_json":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC.json","view_paper":"https://pith.science/paper/GRTXIRD7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.16333&json=true","fetch_graph":"https://pith.science/api/pith-number/GRTXIRD7WJPKAJERWRTCJ4AMQC/graph.json","fetch_events":"https://pith.science/api/pith-number/GRTXIRD7WJPKAJERWRTCJ4AMQC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC/action/storage_attestation","attest_author":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC/action/author_attestation","sign_citation":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC/action/citation_signature","submit_replication":"https://pith.science/pith/GRTXIRD7WJPKAJERWRTCJ4AMQC/action/replication_record"}},"created_at":"2026-07-05T09:00:39.585690+00:00","updated_at":"2026-07-05T09:00:39.585690+00:00"}