{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NUVB3HGM4JM7L2P7OOMJOMKTAA","short_pith_number":"pith:NUVB3HGM","schema_version":"1.0","canonical_sha256":"6d2a1d9ccce259f5e9ff73989731530038e234957260aa86f5b34578ba4221ee","source":{"kind":"arxiv","id":"2407.09499","version":1},"attestation_state":"computed","paper":{"title":"Self-Consuming Generative Models with Curated Data Provably Optimize Human Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Avishek Joey Bose, Damien Ferbach, Gauthier Gidel, Quentin Bertrand","submitted_at":"2024-06-12T21:28:28Z","abstract_excerpt":"The rapid progress in generative models has resulted in impressive leaps in generation quality, blurring the lines between synthetic and real data. Web-scale datasets are now prone to the inevitable contamination by synthetic data, directly impacting the training of future generated models. Already, some theoretical results on self-consuming generative models (a.k.a., iterative retraining) have emerged in the literature, showcasing that either model collapse or stability could be possible depending on the fraction of generated data used at each retraining step. However, in practice, synthetic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09499","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-12T21:28:28Z","cross_cats_sorted":["cs.AI","cs.LG","stat.ML"],"title_canon_sha256":"130aa33e06df19501b0dc4cbe8eb44e7ce56924f740ffdd791f7e2e7e57453a6","abstract_canon_sha256":"5ca0813253c0a9556f6a09ea8a8e22f38d570ba60d859417906e42381ca2758f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:31.066969Z","signature_b64":"R6NgsQxPoN25nnAtP+iXEHNMFPQ0xmF2nB72DwrtQ5Pniu177CrTdjPKnd2Efxiw3xBWzpDeDtLJSKvIIyxEBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d2a1d9ccce259f5e9ff73989731530038e234957260aa86f5b34578ba4221ee","last_reissued_at":"2026-07-05T08:43:31.066494Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:31.066494Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Consuming Generative Models with Curated Data Provably Optimize Human Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Avishek Joey Bose, Damien Ferbach, Gauthier Gidel, Quentin Bertrand","submitted_at":"2024-06-12T21:28:28Z","abstract_excerpt":"The rapid progress in generative models has resulted in impressive leaps in generation quality, blurring the lines between synthetic and real data. Web-scale datasets are now prone to the inevitable contamination by synthetic data, directly impacting the training of future generated models. Already, some theoretical results on self-consuming generative models (a.k.a., iterative retraining) have emerged in the literature, showcasing that either model collapse or stability could be possible depending on the fraction of generated data used at each retraining step. However, in practice, synthetic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09499","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09499","created_at":"2026-07-05T08:43:31.066557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09499v1","created_at":"2026-07-05T08:43:31.066557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09499","created_at":"2026-07-05T08:43:31.066557+00:00"},{"alias_kind":"pith_short_12","alias_value":"NUVB3HGM4JM7","created_at":"2026-07-05T08:43:31.066557+00:00"},{"alias_kind":"pith_short_16","alias_value":"NUVB3HGM4JM7L2P7","created_at":"2026-07-05T08:43:31.066557+00:00"},{"alias_kind":"pith_short_8","alias_value":"NUVB3HGM","created_at":"2026-07-05T08:43:31.066557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28438","citing_title":"When AI Reviews Its Own Code: Recursive Self-Training Collapse in Code LLMs","ref_index":155,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11170","citing_title":"Unlearning with Asymmetric Sources: Improved Unlearning-Utility Trade-off with Public Data","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07724","citing_title":"Curated Synthetic Data Doesn't Have to Collapse: A Theoretical Study of Generative Retraining with Pluralistic Preferences","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA","json":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA.json","graph_json":"https://pith.science/api/pith-number/NUVB3HGM4JM7L2P7OOMJOMKTAA/graph.json","events_json":"https://pith.science/api/pith-number/NUVB3HGM4JM7L2P7OOMJOMKTAA/events.json","paper":"https://pith.science/paper/NUVB3HGM"},"agent_actions":{"view_html":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA","download_json":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA.json","view_paper":"https://pith.science/paper/NUVB3HGM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09499&json=true","fetch_graph":"https://pith.science/api/pith-number/NUVB3HGM4JM7L2P7OOMJOMKTAA/graph.json","fetch_events":"https://pith.science/api/pith-number/NUVB3HGM4JM7L2P7OOMJOMKTAA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA/action/storage_attestation","attest_author":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA/action/author_attestation","sign_citation":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA/action/citation_signature","submit_replication":"https://pith.science/pith/NUVB3HGM4JM7L2P7OOMJOMKTAA/action/replication_record"}},"created_at":"2026-07-05T08:43:31.066557+00:00","updated_at":"2026-07-05T08:43:31.066557+00:00"}