{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AGHYIGJLIY2ILL6AW3BWCICIKM","short_pith_number":"pith:AGHYIGJL","schema_version":"1.0","canonical_sha256":"018f84192b463485afc0b6c361204853257824c22bb763b92811724950e061ab","source":{"kind":"arxiv","id":"2405.09597","version":3},"attestation_state":"computed","paper":{"title":"When AI Eats Itself: On the Caveats of AI Autophagy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Carola-Bibiane Sch\\\"onlieb, Fadong Shi, Guang Yang, Javier Del Ser, Jiahao Huang, Mike Roberts, Sheng Zhang, Xiaodan Xing, Yang Nan, Yingying Fang, Yinzhe Wu","submitted_at":"2024-05-15T13:50:23Z","abstract_excerpt":"Generative Artificial Intelligence (AI) technologies and large models are producing realistic outputs across various domains, such as images, text, speech, and music. Creating these advanced generative models requires significant resources, particularly large and high-quality datasets. To minimise training expenses, many algorithm developers use data created by the models themselves as a cost-effective training solution. However, not all synthetic data effectively improve model performance, necessitating a strategic balance in the use of real versus synthetic data to optimise outcomes. Current"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.09597","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-15T13:50:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d420f56bdfcd5ab16da2fd03b45fd54d2fec1c82ddf8ac88b95dff5bd3980692","abstract_canon_sha256":"f66b593995afbf6d3eadba906e3be58623e9263e942360c9074cdd006e5ec094"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:44.182505Z","signature_b64":"FaRHkgWR7+P6bZhnm+IidkI3zszWN5K0q9dX1JW7elaO+XSPPFDi8cPTEIemC65kcxwtWcxUjDOJVQBuhgsSAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"018f84192b463485afc0b6c361204853257824c22bb763b92811724950e061ab","last_reissued_at":"2026-07-05T09:32:44.181966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:44.181966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When AI Eats Itself: On the Caveats of AI Autophagy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Carola-Bibiane Sch\\\"onlieb, Fadong Shi, Guang Yang, Javier Del Ser, Jiahao Huang, Mike Roberts, Sheng Zhang, Xiaodan Xing, Yang Nan, Yingying Fang, Yinzhe Wu","submitted_at":"2024-05-15T13:50:23Z","abstract_excerpt":"Generative Artificial Intelligence (AI) technologies and large models are producing realistic outputs across various domains, such as images, text, speech, and music. Creating these advanced generative models requires significant resources, particularly large and high-quality datasets. To minimise training expenses, many algorithm developers use data created by the models themselves as a cost-effective training solution. However, not all synthetic data effectively improve model performance, necessitating a strategic balance in the use of real versus synthetic data to optimise outcomes. Current"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.09597","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.09597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.09597","created_at":"2026-07-05T09:32:44.182034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.09597v3","created_at":"2026-07-05T09:32:44.182034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.09597","created_at":"2026-07-05T09:32:44.182034+00:00"},{"alias_kind":"pith_short_12","alias_value":"AGHYIGJLIY2I","created_at":"2026-07-05T09:32:44.182034+00:00"},{"alias_kind":"pith_short_16","alias_value":"AGHYIGJLIY2ILL6A","created_at":"2026-07-05T09:32:44.182034+00:00"},{"alias_kind":"pith_short_8","alias_value":"AGHYIGJL","created_at":"2026-07-05T09:32:44.182034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.01976","citing_title":"A Comprehensive Survey on Network Traffic Synthesis: From Statistical Models to Deep Learning","ref_index":168,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM","json":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM.json","graph_json":"https://pith.science/api/pith-number/AGHYIGJLIY2ILL6AW3BWCICIKM/graph.json","events_json":"https://pith.science/api/pith-number/AGHYIGJLIY2ILL6AW3BWCICIKM/events.json","paper":"https://pith.science/paper/AGHYIGJL"},"agent_actions":{"view_html":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM","download_json":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM.json","view_paper":"https://pith.science/paper/AGHYIGJL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.09597&json=true","fetch_graph":"https://pith.science/api/pith-number/AGHYIGJLIY2ILL6AW3BWCICIKM/graph.json","fetch_events":"https://pith.science/api/pith-number/AGHYIGJLIY2ILL6AW3BWCICIKM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM/action/storage_attestation","attest_author":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM/action/author_attestation","sign_citation":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM/action/citation_signature","submit_replication":"https://pith.science/pith/AGHYIGJLIY2ILL6AW3BWCICIKM/action/replication_record"}},"created_at":"2026-07-05T09:32:44.182034+00:00","updated_at":"2026-07-05T09:32:44.182034+00:00"}