{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P7U2LGARS3QEZICSNO5E4OMUKW","short_pith_number":"pith:P7U2LGAR","schema_version":"1.0","canonical_sha256":"7fe9a5981196e04ca0526bba4e399455b759a0d17234d4855bd7e749aea9a955","source":{"kind":"arxiv","id":"2410.21717","version":1},"attestation_state":"computed","paper":{"title":"Generating Realistic Tabular Data with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dang Nguyen, Kien Do, Sunil Gupta, Svetha Venkatesh, Thin Nguyen","submitted_at":"2024-10-29T04:14:32Z","abstract_excerpt":"While most generative models show achievements in image data generation, few are developed for tabular data generation. Recently, due to success of large language models (LLM) in diverse tasks, they have also been used for tabular data generation. However, these methods do not capture the correct correlation between the features and the target variable, hindering their applications in downstream predictive tasks. To address this problem, we propose a LLM-based method with three important improvements to correctly capture the ground-truth feature-class correlation in the real data. First, we pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21717","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-29T04:14:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4d01b0d73068139bd6d575a11bf961b934e25be02b07f0b10ccbbc30d82e8748","abstract_canon_sha256":"636eb945ab4beaa0aa6b25ef05f5ad38f12519106ab75ae204d54dc43f4dc883"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:32.190080Z","signature_b64":"1FsrKqVzQkNA6mXIj4ztPSHpE78ZwGM8sJY44zOk5BkAAQWd0mYNkD7ufpTgiMaQ4gcDlSPMC/snQ/htKLGNDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fe9a5981196e04ca0526bba4e399455b759a0d17234d4855bd7e749aea9a955","last_reissued_at":"2026-07-05T09:27:32.189650Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:32.189650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generating Realistic Tabular Data with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dang Nguyen, Kien Do, Sunil Gupta, Svetha Venkatesh, Thin Nguyen","submitted_at":"2024-10-29T04:14:32Z","abstract_excerpt":"While most generative models show achievements in image data generation, few are developed for tabular data generation. Recently, due to success of large language models (LLM) in diverse tasks, they have also been used for tabular data generation. However, these methods do not capture the correct correlation between the features and the target variable, hindering their applications in downstream predictive tasks. To address this problem, we propose a LLM-based method with three important improvements to correctly capture the ground-truth feature-class correlation in the real data. First, we pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21717","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21717/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21717","created_at":"2026-07-05T09:27:32.189712+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21717v1","created_at":"2026-07-05T09:27:32.189712+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21717","created_at":"2026-07-05T09:27:32.189712+00:00"},{"alias_kind":"pith_short_12","alias_value":"P7U2LGARS3QE","created_at":"2026-07-05T09:27:32.189712+00:00"},{"alias_kind":"pith_short_16","alias_value":"P7U2LGARS3QEZICS","created_at":"2026-07-05T09:27:32.189712+00:00"},{"alias_kind":"pith_short_8","alias_value":"P7U2LGAR","created_at":"2026-07-05T09:27:32.189712+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10088","citing_title":"Towards High Supervised Learning Utility Training Data Generation: Data Pruning and Column Reordering","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW","json":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW.json","graph_json":"https://pith.science/api/pith-number/P7U2LGARS3QEZICSNO5E4OMUKW/graph.json","events_json":"https://pith.science/api/pith-number/P7U2LGARS3QEZICSNO5E4OMUKW/events.json","paper":"https://pith.science/paper/P7U2LGAR"},"agent_actions":{"view_html":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW","download_json":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW.json","view_paper":"https://pith.science/paper/P7U2LGAR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21717&json=true","fetch_graph":"https://pith.science/api/pith-number/P7U2LGARS3QEZICSNO5E4OMUKW/graph.json","fetch_events":"https://pith.science/api/pith-number/P7U2LGARS3QEZICSNO5E4OMUKW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW/action/storage_attestation","attest_author":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW/action/author_attestation","sign_citation":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW/action/citation_signature","submit_replication":"https://pith.science/pith/P7U2LGARS3QEZICSNO5E4OMUKW/action/replication_record"}},"created_at":"2026-07-05T09:27:32.189712+00:00","updated_at":"2026-07-05T09:27:32.189712+00:00"}