{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:N4HPW5HBKFNYT4RMMPIMXTDFYF","short_pith_number":"pith:N4HPW5HB","schema_version":"1.0","canonical_sha256":"6f0efb74e1515b89f22c63d0cbcc65c15f3d56a1edef6688efaba3a224a59c26","source":{"kind":"arxiv","id":"2307.01850","version":1},"attestation_state":"computed","paper":{"title":"Self-Consuming Generative Models Go MAD","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Ahmed Imtiaz Humayun, Ali Siahkoohi, Daniel LeJeune, Hossein Babaei, Josue Casco-Rodriguez, Lorenzo Luzi, Richard G. Baraniuk, Sina Alemohammad","submitted_at":"2023-07-04T17:59:31Z","abstract_excerpt":"Seismic advances in generative AI algorithms for imagery, text, and other data types has led to the temptation to use synthetic data to train next-generation models. Repeating this process creates an autophagous (self-consuming) loop whose properties are poorly understood. We conduct a thorough analytical and empirical analysis using state-of-the-art generative image models of three families of autophagous loops that differ in how fixed or fresh real training data is available through the generations of training and in whether the samples from previous generation models have been biased to tra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.01850","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-07-04T17:59:31Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"373fb00532ea505d5a7e95365e03019d74c3c3f8d7e686c82c6c23bce619ec46","abstract_canon_sha256":"ccd7e716221d48eb474351da99cdc8133a91441e99e3362e2eac365777891728"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:28:01.717933Z","signature_b64":"WANASh+w06n1sYmju7IJv7EO1OInSkuDtQQ3n5lK6uEfugV5TGSnSPeBr8ybm2aU1KYNGgfLu4c+c3wsUfSlCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f0efb74e1515b89f22c63d0cbcc65c15f3d56a1edef6688efaba3a224a59c26","last_reissued_at":"2026-07-05T06:28:01.717482Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:28:01.717482Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Consuming Generative Models Go MAD","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Ahmed Imtiaz Humayun, Ali Siahkoohi, Daniel LeJeune, Hossein Babaei, Josue Casco-Rodriguez, Lorenzo Luzi, Richard G. Baraniuk, Sina Alemohammad","submitted_at":"2023-07-04T17:59:31Z","abstract_excerpt":"Seismic advances in generative AI algorithms for imagery, text, and other data types has led to the temptation to use synthetic data to train next-generation models. Repeating this process creates an autophagous (self-consuming) loop whose properties are poorly understood. We conduct a thorough analytical and empirical analysis using state-of-the-art generative image models of three families of autophagous loops that differ in how fixed or fresh real training data is available through the generations of training and in whether the samples from previous generation models have been biased to tra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.01850","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.01850/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.01850","created_at":"2026-07-05T06:28:01.717540+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.01850v1","created_at":"2026-07-05T06:28:01.717540+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.01850","created_at":"2026-07-05T06:28:01.717540+00:00"},{"alias_kind":"pith_short_12","alias_value":"N4HPW5HBKFNY","created_at":"2026-07-05T06:28:01.717540+00:00"},{"alias_kind":"pith_short_16","alias_value":"N4HPW5HBKFNYT4RM","created_at":"2026-07-05T06:28:01.717540+00:00"},{"alias_kind":"pith_short_8","alias_value":"N4HPW5HB","created_at":"2026-07-05T06:28:01.717540+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18288","citing_title":"A Knowledge Theory of Capital:The Value of Natural and Artificial Intelligence, Volume 1","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00334","citing_title":"Isolating LLM Lexical Bias: A Curation-Free Triangulated Metric for Preference-Stage Learning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18288","citing_title":"A Knowledge Theory of Capital:The Value of Natural and Artificial Intelligence, Volume 1","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10113","citing_title":"Emotion Profiling in LLM-Based Literary Translation: Systematic Shifts Across MT and Post-Editing","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2503.08223","citing_title":"Will LLMs Scaling Hit the Wall? Breaking Barriers via Distributed Resources on Massive Edge Devices","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20602","citing_title":"Self-Training Doesn't Flatten Language -- It Restructures It: Surface Markers Amplify While Deep Syntax Dies","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17193","citing_title":"Multi-LLM Systems Exhibit Robust Semantic Collapse","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06024","citing_title":"On Inverse Problems, Parameter Estimation, and Domain Generalization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.03933","citing_title":"Losing our Tail, Again: (Un)Natural Selection & Multilingual LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08115","citing_title":"Alice v1: Distillation-Enhanced Video Generation Surpassing Closed-Source Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26965","citing_title":"The Impact of AI-Generated Text on the Internet","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15786","citing_title":"Filter Babel: The Challenge of Synthetic Media to Authenticity and Common Ground in AI-Mediated Communication","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02236","citing_title":"Perturbation Dose Responses in Recursive LLM Loops: Raw Switching, Stochastic Floors, and Persistent Escape under Append, Replace, and Dialog Updates","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF","json":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF.json","graph_json":"https://pith.science/api/pith-number/N4HPW5HBKFNYT4RMMPIMXTDFYF/graph.json","events_json":"https://pith.science/api/pith-number/N4HPW5HBKFNYT4RMMPIMXTDFYF/events.json","paper":"https://pith.science/paper/N4HPW5HB"},"agent_actions":{"view_html":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF","download_json":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF.json","view_paper":"https://pith.science/paper/N4HPW5HB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.01850&json=true","fetch_graph":"https://pith.science/api/pith-number/N4HPW5HBKFNYT4RMMPIMXTDFYF/graph.json","fetch_events":"https://pith.science/api/pith-number/N4HPW5HBKFNYT4RMMPIMXTDFYF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF/action/storage_attestation","attest_author":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF/action/author_attestation","sign_citation":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF/action/citation_signature","submit_replication":"https://pith.science/pith/N4HPW5HBKFNYT4RMMPIMXTDFYF/action/replication_record"}},"created_at":"2026-07-05T06:28:01.717540+00:00","updated_at":"2026-07-05T06:28:01.717540+00:00"}