{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QXYOWMNLRFPGBLZDUEMGHFDBRD","short_pith_number":"pith:QXYOWMNL","schema_version":"1.0","canonical_sha256":"85f0eb31ab895e60af23a11863946188ff6c809f390793f1649f8460d98ed0da","source":{"kind":"arxiv","id":"2305.06090","version":1},"attestation_state":"computed","paper":{"title":"XTab: Cross-table Pretraining for Tabular Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bingzhao Zhu, George Karypis, Mahsa Shoaran, Mu Li, Nick Erickson, Xingjian Shi","submitted_at":"2023-05-10T12:17:52Z","abstract_excerpt":"The success of self-supervised learning in computer vision and natural language processing has motivated pretraining methods on tabular data. However, most existing tabular self-supervised learning models fail to leverage information across multiple data tables and cannot generalize to new tables. In this work, we introduce XTab, a framework for cross-table pretraining of tabular transformers on datasets from various domains. We address the challenge of inconsistent column types and quantities among tables by utilizing independent featurizers and using federated learning to pretrain the shared"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.06090","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-05-10T12:17:52Z","cross_cats_sorted":[],"title_canon_sha256":"3c2ea4c8a38e3715e9de5713bea1ebed191df82195ad195dcc2917dda9a62e6b","abstract_canon_sha256":"eb889e34411d2ac9e95f2991c8413e48d72180fd4395166da9892158b3f8d898"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:08:59.604894Z","signature_b64":"AT17sgvTC2BnOhV096MhaVmzekQsG7Gd98hfFoXtj+G6dkzSf9h1HxWgq6knvziT2rell8ZEmbuKrtRjFvD0Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85f0eb31ab895e60af23a11863946188ff6c809f390793f1649f8460d98ed0da","last_reissued_at":"2026-07-05T06:08:59.604476Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:08:59.604476Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XTab: Cross-table Pretraining for Tabular Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bingzhao Zhu, George Karypis, Mahsa Shoaran, Mu Li, Nick Erickson, Xingjian Shi","submitted_at":"2023-05-10T12:17:52Z","abstract_excerpt":"The success of self-supervised learning in computer vision and natural language processing has motivated pretraining methods on tabular data. However, most existing tabular self-supervised learning models fail to leverage information across multiple data tables and cannot generalize to new tables. In this work, we introduce XTab, a framework for cross-table pretraining of tabular transformers on datasets from various domains. We address the challenge of inconsistent column types and quantities among tables by utilizing independent featurizers and using federated learning to pretrain the shared"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.06090","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.06090/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.06090","created_at":"2026-07-05T06:08:59.604532+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.06090v1","created_at":"2026-07-05T06:08:59.604532+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.06090","created_at":"2026-07-05T06:08:59.604532+00:00"},{"alias_kind":"pith_short_12","alias_value":"QXYOWMNLRFPG","created_at":"2026-07-05T06:08:59.604532+00:00"},{"alias_kind":"pith_short_16","alias_value":"QXYOWMNLRFPGBLZD","created_at":"2026-07-05T06:08:59.604532+00:00"},{"alias_kind":"pith_short_8","alias_value":"QXYOWMNL","created_at":"2026-07-05T06:08:59.604532+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2403.20208","citing_title":"Unlock the Potential of Large Language Models for Predictive Tabular Tasks in Data Science with Table-Specific Pretraining","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04962","citing_title":"TabEmbed: Benchmarking and Learning Generalist Embeddings for Tabular Understanding","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD","json":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD.json","graph_json":"https://pith.science/api/pith-number/QXYOWMNLRFPGBLZDUEMGHFDBRD/graph.json","events_json":"https://pith.science/api/pith-number/QXYOWMNLRFPGBLZDUEMGHFDBRD/events.json","paper":"https://pith.science/paper/QXYOWMNL"},"agent_actions":{"view_html":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD","download_json":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD.json","view_paper":"https://pith.science/paper/QXYOWMNL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.06090&json=true","fetch_graph":"https://pith.science/api/pith-number/QXYOWMNLRFPGBLZDUEMGHFDBRD/graph.json","fetch_events":"https://pith.science/api/pith-number/QXYOWMNLRFPGBLZDUEMGHFDBRD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD/action/storage_attestation","attest_author":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD/action/author_attestation","sign_citation":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD/action/citation_signature","submit_replication":"https://pith.science/pith/QXYOWMNLRFPGBLZDUEMGHFDBRD/action/replication_record"}},"created_at":"2026-07-05T06:08:59.604532+00:00","updated_at":"2026-07-05T06:08:59.604532+00:00"}