{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:KAMND5B2RSHRHTM46CWV6IJUCY","short_pith_number":"pith:KAMND5B2","schema_version":"1.0","canonical_sha256":"5018d1f43a8c8f13cd9cf0ad5f2134161e3911b8ffdaf71e90b9c1e382a8fae9","source":{"kind":"arxiv","id":"2210.04018","version":4},"attestation_state":"computed","paper":{"title":"STaSy: Score-based Tabular data Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DB"],"primary_cat":"cs.LG","authors_text":"Chaejeong Lee, Jayoung Kim, Noseong Park","submitted_at":"2022-10-08T13:09:51Z","abstract_excerpt":"Tabular data synthesis is a long-standing research topic in machine learning. Many different methods have been proposed over the past decades, ranging from statistical methods to deep generative methods. However, it has not always been successful due to the complicated nature of real-world tabular data. In this paper, we present a new model named Score-based Tabular data Synthesis (STaSy) and its training strategy based on the paradigm of score-based generative modeling. Despite the fact that score-based generative models have resolved many issues in generative models, there still exists room "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.04018","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-08T13:09:51Z","cross_cats_sorted":["cs.AI","cs.DB"],"title_canon_sha256":"f9c2b5f2d270346f0628763a27c594bdb2f1e724087563a58a125f4a6b76afe0","abstract_canon_sha256":"0bb9ea412f953e1484e137fa7d35f9ce7f99086d9795a2f7e88f96d104c21c12"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:37.085922Z","signature_b64":"9kEfjZ1uz95EGlj91FGgdxgjTLD+K/Cr2HnsUf9VGUu++QwB05RH7NACbZjlb5PGjBE4jTGbIG/B3H54m0MBCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5018d1f43a8c8f13cd9cf0ad5f2134161e3911b8ffdaf71e90b9c1e382a8fae9","last_reissued_at":"2026-07-05T06:14:37.085419Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:37.085419Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"STaSy: Score-based Tabular data Synthesis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.DB"],"primary_cat":"cs.LG","authors_text":"Chaejeong Lee, Jayoung Kim, Noseong Park","submitted_at":"2022-10-08T13:09:51Z","abstract_excerpt":"Tabular data synthesis is a long-standing research topic in machine learning. Many different methods have been proposed over the past decades, ranging from statistical methods to deep generative methods. However, it has not always been successful due to the complicated nature of real-world tabular data. In this paper, we present a new model named Score-based Tabular data Synthesis (STaSy) and its training strategy based on the paradigm of score-based generative modeling. Despite the fact that score-based generative models have resolved many issues in generative models, there still exists room "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.04018","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.04018/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.04018","created_at":"2026-07-05T06:14:37.085480+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.04018v4","created_at":"2026-07-05T06:14:37.085480+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.04018","created_at":"2026-07-05T06:14:37.085480+00:00"},{"alias_kind":"pith_short_12","alias_value":"KAMND5B2RSHR","created_at":"2026-07-05T06:14:37.085480+00:00"},{"alias_kind":"pith_short_16","alias_value":"KAMND5B2RSHRHTM4","created_at":"2026-07-05T06:14:37.085480+00:00"},{"alias_kind":"pith_short_8","alias_value":"KAMND5B2","created_at":"2026-07-05T06:14:37.085480+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.02161","citing_title":"LLM-TabLogic: Preserving Inter-Column Logical Relationships in Synthetic Tabular Data via Prompt-Guided Latent Diffusion","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19185","citing_title":"MIDST Challenge at SaTML 2025: Membership Inference over Diffusion-models-based Synthetic Tabular data","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06261","citing_title":"Inference-Time Refinement Closes the Synthetic-Real Gap in Tabular Diffusion","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18966","citing_title":"Self-Improving Tabular Language Models via Iterative Reward-Guided Post-Training","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04911","citing_title":"Breaking the Quality-Privacy Tradeoff in Tabular Data Generation via In-Context Learning","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY","json":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY.json","graph_json":"https://pith.science/api/pith-number/KAMND5B2RSHRHTM46CWV6IJUCY/graph.json","events_json":"https://pith.science/api/pith-number/KAMND5B2RSHRHTM46CWV6IJUCY/events.json","paper":"https://pith.science/paper/KAMND5B2"},"agent_actions":{"view_html":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY","download_json":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY.json","view_paper":"https://pith.science/paper/KAMND5B2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.04018&json=true","fetch_graph":"https://pith.science/api/pith-number/KAMND5B2RSHRHTM46CWV6IJUCY/graph.json","fetch_events":"https://pith.science/api/pith-number/KAMND5B2RSHRHTM46CWV6IJUCY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY/action/storage_attestation","attest_author":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY/action/author_attestation","sign_citation":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY/action/citation_signature","submit_replication":"https://pith.science/pith/KAMND5B2RSHRHTM46CWV6IJUCY/action/replication_record"}},"created_at":"2026-07-05T06:14:37.085480+00:00","updated_at":"2026-07-05T06:14:37.085480+00:00"}