{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5I5HETH653R6VYB5HZTSBJH6Y7","short_pith_number":"pith:5I5HETH6","schema_version":"1.0","canonical_sha256":"ea3a724cfeeee3eae03d3e6720a4fec7e626e2d96131c2936ac9a02249609bfc","source":{"kind":"arxiv","id":"2401.02524","version":2},"attestation_state":"computed","paper":{"title":"Comprehensive Exploration of Synthetic Data Generation: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Andr\\'e Bauer, Ian Foster, Kyle Chard, Mark Leznik, Michael Stenger, Robert Leppich, Samuel Kounev, Simon Trapp","submitted_at":"2024-01-04T20:23:51Z","abstract_excerpt":"Recent years have witnessed a surge in the popularity of Machine Learning (ML), applied across diverse domains. However, progress is impeded by the scarcity of training data due to expensive acquisition and privacy legislation. Synthetic data emerges as a solution, but the abundance of released models and limited overview literature pose challenges for decision-making. This work surveys 417 Synthetic Data Generation (SDG) models over the last decade, providing a comprehensive overview of model types, functionality, and improvements. Common attributes are identified, leading to a classification"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.02524","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-04T20:23:51Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"da199a877880a109f0041efa701de65e934c7895b063c7e64ef3245e4e3a7daf","abstract_canon_sha256":"bc2237eb6555d8782944e0792f193b5c05f417417603df54026acff9fb80c93f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:26.125466Z","signature_b64":"cV/rTbChb/q7zzQeYiu9kzzF8vfz1yBFH9zbmddv0ZjmDFnthGVarQ/T2mHXj30xoBDlgtFtwgvjadPybHQBBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea3a724cfeeee3eae03d3e6720a4fec7e626e2d96131c2936ac9a02249609bfc","last_reissued_at":"2026-07-05T07:40:26.125034Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:26.125034Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comprehensive Exploration of Synthetic Data Generation: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Andr\\'e Bauer, Ian Foster, Kyle Chard, Mark Leznik, Michael Stenger, Robert Leppich, Samuel Kounev, Simon Trapp","submitted_at":"2024-01-04T20:23:51Z","abstract_excerpt":"Recent years have witnessed a surge in the popularity of Machine Learning (ML), applied across diverse domains. However, progress is impeded by the scarcity of training data due to expensive acquisition and privacy legislation. Synthetic data emerges as a solution, but the abundance of released models and limited overview literature pose challenges for decision-making. This work surveys 417 Synthetic Data Generation (SDG) models over the last decade, providing a comprehensive overview of model types, functionality, and improvements. Common attributes are identified, leading to a classification"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.02524","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02524/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.02524","created_at":"2026-07-05T07:40:26.125090+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.02524v2","created_at":"2026-07-05T07:40:26.125090+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02524","created_at":"2026-07-05T07:40:26.125090+00:00"},{"alias_kind":"pith_short_12","alias_value":"5I5HETH653R6","created_at":"2026-07-05T07:40:26.125090+00:00"},{"alias_kind":"pith_short_16","alias_value":"5I5HETH653R6VYB5","created_at":"2026-07-05T07:40:26.125090+00:00"},{"alias_kind":"pith_short_8","alias_value":"5I5HETH6","created_at":"2026-07-05T07:40:26.125090+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06334","citing_title":"Quantifying the Privacy of Counterfactuals by Leveraging Membership Inference Attacks Against Synthetic Data","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02947","citing_title":"Quality Degradation Attack in Synthetic Data","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18499","citing_title":"PuckTrick: A Library for Making Synthetic Data More Realistic","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21600","citing_title":"Robust Spectral Watermark for Synthetic Tabular Data","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2406.20094","citing_title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17803","citing_title":"Adversarial Arena: Crowdsourcing Data Generation through Interactive Competition","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7","json":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7.json","graph_json":"https://pith.science/api/pith-number/5I5HETH653R6VYB5HZTSBJH6Y7/graph.json","events_json":"https://pith.science/api/pith-number/5I5HETH653R6VYB5HZTSBJH6Y7/events.json","paper":"https://pith.science/paper/5I5HETH6"},"agent_actions":{"view_html":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7","download_json":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7.json","view_paper":"https://pith.science/paper/5I5HETH6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.02524&json=true","fetch_graph":"https://pith.science/api/pith-number/5I5HETH653R6VYB5HZTSBJH6Y7/graph.json","fetch_events":"https://pith.science/api/pith-number/5I5HETH653R6VYB5HZTSBJH6Y7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7/action/storage_attestation","attest_author":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7/action/author_attestation","sign_citation":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7/action/citation_signature","submit_replication":"https://pith.science/pith/5I5HETH653R6VYB5HZTSBJH6Y7/action/replication_record"}},"created_at":"2026-07-05T07:40:26.125090+00:00","updated_at":"2026-07-05T07:40:26.125090+00:00"}