{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B5YA6CSKJIVSC3FRGGK7AXWB74","short_pith_number":"pith:B5YA6CSK","schema_version":"1.0","canonical_sha256":"0f700f0a4a4a2b216cb13195f05ec1ff26af49042d750537e6151233b4920cd5","source":{"kind":"arxiv","id":"2504.16506","version":3},"attestation_state":"computed","paper":{"title":"A Comprehensive Survey of Synthetic Tabular Data Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Mengnan Du, Ruxue Shi, Xin Wang, Xu Shen, Yi Chang, Yili Wang","submitted_at":"2025-04-23T08:33:34Z","abstract_excerpt":"Tabular data is one of the most prevalent and important data formats in real-world applications such as healthcare, finance, and education. However, its effective use in machine learning is often constrained by data scarcity, privacy concerns, and class imbalance. Synthetic tabular data generation has emerged as a powerful solution, leveraging generative models to learn underlying data distributions and produce realistic, privacy-preserving samples. Although this area has seen growing attention, most existing surveys focus narrowly on specific methods (e.g., GANs or privacy-enhancing technique"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16506","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-23T08:33:34Z","cross_cats_sorted":[],"title_canon_sha256":"4babbf1dbd99fe0f038606262172431bffa817d9a1aed019ba516116da2915d3","abstract_canon_sha256":"cd41cc0a31c3e79197b44fb49cf2c442d72ed30fcdab400abfd6ca3c138a77ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:28.977202Z","signature_b64":"IMmAVUoPkgzE6nPy0WRy3lI7x3T6kraTUUvx4XH3m3O20srJ5Bw+n/oNQz2pqIVbhLOfFaQTQpOfuDtfKLfFCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f700f0a4a4a2b216cb13195f05ec1ff26af49042d750537e6151233b4920cd5","last_reissued_at":"2026-07-05T11:38:28.976690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:28.976690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Survey of Synthetic Tabular Data Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Mengnan Du, Ruxue Shi, Xin Wang, Xu Shen, Yi Chang, Yili Wang","submitted_at":"2025-04-23T08:33:34Z","abstract_excerpt":"Tabular data is one of the most prevalent and important data formats in real-world applications such as healthcare, finance, and education. However, its effective use in machine learning is often constrained by data scarcity, privacy concerns, and class imbalance. Synthetic tabular data generation has emerged as a powerful solution, leveraging generative models to learn underlying data distributions and produce realistic, privacy-preserving samples. Although this area has seen growing attention, most existing surveys focus narrowly on specific methods (e.g., GANs or privacy-enhancing technique"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16506","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16506/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16506","created_at":"2026-07-05T11:38:28.976753+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16506v3","created_at":"2026-07-05T11:38:28.976753+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16506","created_at":"2026-07-05T11:38:28.976753+00:00"},{"alias_kind":"pith_short_12","alias_value":"B5YA6CSKJIVS","created_at":"2026-07-05T11:38:28.976753+00:00"},{"alias_kind":"pith_short_16","alias_value":"B5YA6CSKJIVSC3FR","created_at":"2026-07-05T11:38:28.976753+00:00"},{"alias_kind":"pith_short_8","alias_value":"B5YA6CSK","created_at":"2026-07-05T11:38:28.976753+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06133","citing_title":"Property-Driven Synthetic Data Engineering for Data-Scarce Software Systems: Reflections from the Breast Cancer Domain","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31904","citing_title":"Sequential RC-TGAN: Generating Relational Time Series with Spectral Envelope Loss","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2501.15461","citing_title":"Mamba-Based Graph Convolutional Networks: Tackling Over-smoothing with Selective State Space","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17642","citing_title":"TabKDE: Simple and Scalable Tabular Data Generation with Kernel Density Estimates","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01444","citing_title":"Autoregressive Synthesis of Sparse and Semi-Structured Mixed-Type Data","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09424","citing_title":"Tabular Foundation Model for Generative Modelling","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24368","citing_title":"SAGE: Sparse Adaptive Guidance for Dependency-Aware Tabular Data Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05752","citing_title":"Generative AI-Based Monte Carlo Simulation for Method Evaluation Using Synthetic Multilevel Data","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18266","citing_title":"Enhancing Tabular Anomaly Detection via Pseudo-Label-Guided Generation","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74","json":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74.json","graph_json":"https://pith.science/api/pith-number/B5YA6CSKJIVSC3FRGGK7AXWB74/graph.json","events_json":"https://pith.science/api/pith-number/B5YA6CSKJIVSC3FRGGK7AXWB74/events.json","paper":"https://pith.science/paper/B5YA6CSK"},"agent_actions":{"view_html":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74","download_json":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74.json","view_paper":"https://pith.science/paper/B5YA6CSK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16506&json=true","fetch_graph":"https://pith.science/api/pith-number/B5YA6CSKJIVSC3FRGGK7AXWB74/graph.json","fetch_events":"https://pith.science/api/pith-number/B5YA6CSKJIVSC3FRGGK7AXWB74/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74/action/storage_attestation","attest_author":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74/action/author_attestation","sign_citation":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74/action/citation_signature","submit_replication":"https://pith.science/pith/B5YA6CSKJIVSC3FRGGK7AXWB74/action/replication_record"}},"created_at":"2026-07-05T11:38:28.976753+00:00","updated_at":"2026-07-05T11:38:28.976753+00:00"}