{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PI6LQ2YMKLJKPOBBGFLG5W2ZKO","short_pith_number":"pith:PI6LQ2YM","schema_version":"1.0","canonical_sha256":"7a3cb86b0c52d2a7b82131566edb5953aaca41de891932b728e7f2ff82c08921","source":{"kind":"arxiv","id":"2310.12746","version":3},"attestation_state":"computed","paper":{"title":"TabuLa: Harnessing Language Models for Tabular Data Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lydia Chen, Robert Birke, Zilong Zhao","submitted_at":"2023-10-19T13:50:56Z","abstract_excerpt":"Tabular data synthesis is crucial for addressing privacy and security concerns in industries reliant on tabular data. While recent advancements adopt large language models (LLMs) for realistic tabular data generation, their long training times and limited reusability hinder practical applications. In this paper, we propose Tabula, a tabular data synthesizer that leverages the structure of LLM. Unlike state-of-the-art (SOTA) LLM-based tabular data synthesizers that rely on pre-trained LLMs, Tabula discards the pre-trained weights originally designed for natural language tasks, focusing instead "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.12746","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-19T13:50:56Z","cross_cats_sorted":[],"title_canon_sha256":"3ac1975b9976d73a2a8af3caf6f616d9cadd83c0dea998709468943e3d34460e","abstract_canon_sha256":"026168efa3166f0b2f7ccc420af457d06e0dee2eaaef4ee972e1cd6d006c8dbf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:14.743132Z","signature_b64":"21HRzeh9K/aeHoR/JQ9tBnmCtfjCcjb0mGxwXzKhxEMFTu34Hu9PMjFm1QtbmzyKq5+ac6qCZ/kQO/h9eU3cDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a3cb86b0c52d2a7b82131566edb5953aaca41de891932b728e7f2ff82c08921","last_reissued_at":"2026-07-05T10:15:14.742635Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:14.742635Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TabuLa: Harnessing Language Models for Tabular Data Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lydia Chen, Robert Birke, Zilong Zhao","submitted_at":"2023-10-19T13:50:56Z","abstract_excerpt":"Tabular data synthesis is crucial for addressing privacy and security concerns in industries reliant on tabular data. While recent advancements adopt large language models (LLMs) for realistic tabular data generation, their long training times and limited reusability hinder practical applications. In this paper, we propose Tabula, a tabular data synthesizer that leverages the structure of LLM. Unlike state-of-the-art (SOTA) LLM-based tabular data synthesizers that rely on pre-trained LLMs, Tabula discards the pre-trained weights originally designed for natural language tasks, focusing instead "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.12746","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.12746/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.12746","created_at":"2026-07-05T10:15:14.742694+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.12746v3","created_at":"2026-07-05T10:15:14.742694+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.12746","created_at":"2026-07-05T10:15:14.742694+00:00"},{"alias_kind":"pith_short_12","alias_value":"PI6LQ2YMKLJK","created_at":"2026-07-05T10:15:14.742694+00:00"},{"alias_kind":"pith_short_16","alias_value":"PI6LQ2YMKLJKPOBB","created_at":"2026-07-05T10:15:14.742694+00:00"},{"alias_kind":"pith_short_8","alias_value":"PI6LQ2YM","created_at":"2026-07-05T10:15:14.742694+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11961","citing_title":"Categorical Prior Lock-in: Why In-Context Learning Fails for Structured Data","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02161","citing_title":"LLM-TabLogic: Preserving Inter-Column Logical Relationships in Synthetic Tabular Data via Prompt-Guided Latent Diffusion","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO","json":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO.json","graph_json":"https://pith.science/api/pith-number/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/graph.json","events_json":"https://pith.science/api/pith-number/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/events.json","paper":"https://pith.science/paper/PI6LQ2YM"},"agent_actions":{"view_html":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO","download_json":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO.json","view_paper":"https://pith.science/paper/PI6LQ2YM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.12746&json=true","fetch_graph":"https://pith.science/api/pith-number/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/graph.json","fetch_events":"https://pith.science/api/pith-number/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/action/storage_attestation","attest_author":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/action/author_attestation","sign_citation":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/action/citation_signature","submit_replication":"https://pith.science/pith/PI6LQ2YMKLJKPOBBGFLG5W2ZKO/action/replication_record"}},"created_at":"2026-07-05T10:15:14.742694+00:00","updated_at":"2026-07-05T10:15:14.742694+00:00"}