{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7XSJ2CHZR7MOQVHQVBSWVZCMLD","short_pith_number":"pith:7XSJ2CHZ","schema_version":"1.0","canonical_sha256":"fde49d08f98fd8e854f0a8656ae44c58fa3804713c8dde1c04d6d0ce55d6c173","source":{"kind":"arxiv","id":"2406.12031","version":2},"attestation_state":"computed","paper":{"title":"Large Scale Transfer Learning for Tabular Data via Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Josh Gardner, Juan C. Perdomo, Ludwig Schmidt","submitted_at":"2024-06-17T18:58:20Z","abstract_excerpt":"Tabular data -- structured, heterogeneous, spreadsheet-style data with rows and columns -- is widely used in practice across many domains. However, while recent foundation models have reduced the need for developing task-specific datasets and predictors in domains such as language modeling and computer vision, this transfer learning paradigm has not had similar impact in the tabular domain. In this work, we seek to narrow this gap and present TabuLa-8B, a language model for tabular prediction. We define a process for extracting a large, high-quality training dataset from the TabLib corpus, pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12031","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-17T18:58:20Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"66f1467228386fb7eec4dd4f98aa466ba7f300f4f175bf2f74cea6b5560593fe","abstract_canon_sha256":"3234e066eced14e42b817f93b929c3d9c52270a5d5b222972a52b1737e489498"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:27.356777Z","signature_b64":"GzS+nIpeUBVy0KNZ7V5kBJGpfL/wUn+USdNOuG4UsWn36OIxmPM/hzMoIO9B5c48upkc3uyplHMWz+AOsAyKCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fde49d08f98fd8e854f0a8656ae44c58fa3804713c8dde1c04d6d0ce55d6c173","last_reissued_at":"2026-07-05T09:38:27.356120Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:27.356120Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Scale Transfer Learning for Tabular Data via Language Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Josh Gardner, Juan C. Perdomo, Ludwig Schmidt","submitted_at":"2024-06-17T18:58:20Z","abstract_excerpt":"Tabular data -- structured, heterogeneous, spreadsheet-style data with rows and columns -- is widely used in practice across many domains. However, while recent foundation models have reduced the need for developing task-specific datasets and predictors in domains such as language modeling and computer vision, this transfer learning paradigm has not had similar impact in the tabular domain. In this work, we seek to narrow this gap and present TabuLa-8B, a language model for tabular prediction. We define a process for extracting a large, high-quality training dataset from the TabLib corpus, pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12031","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12031/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12031","created_at":"2026-07-05T09:38:27.356179+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12031v2","created_at":"2026-07-05T09:38:27.356179+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12031","created_at":"2026-07-05T09:38:27.356179+00:00"},{"alias_kind":"pith_short_12","alias_value":"7XSJ2CHZR7MO","created_at":"2026-07-05T09:38:27.356179+00:00"},{"alias_kind":"pith_short_16","alias_value":"7XSJ2CHZR7MOQVHQ","created_at":"2026-07-05T09:38:27.356179+00:00"},{"alias_kind":"pith_short_8","alias_value":"7XSJ2CHZ","created_at":"2026-07-05T09:38:27.356179+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07399","citing_title":"Automatic, Debiased, and Invariant Counterfactual Generation under General Interventions","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24417","citing_title":"LLMTabBench: Evaluating LLMs on Binary Tabular Classification From Zero to Few Shots","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29371","citing_title":"Semantic insurance pricing with large language models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31272","citing_title":"Algorithmic Recourse of In-Context Learning for Tabular Data","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17034","citing_title":"Privacy Policy Enforcement Guardrails for Data-Sensitive Retrieval-Augmented Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02832","citing_title":"Transfer Learning for Loan Recovery Prediction under Distribution Shifts with Heterogeneous Feature Spaces","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06290","citing_title":"Data Language Models: A New Foundation Model Class for Tabular Data","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD","json":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD.json","graph_json":"https://pith.science/api/pith-number/7XSJ2CHZR7MOQVHQVBSWVZCMLD/graph.json","events_json":"https://pith.science/api/pith-number/7XSJ2CHZR7MOQVHQVBSWVZCMLD/events.json","paper":"https://pith.science/paper/7XSJ2CHZ"},"agent_actions":{"view_html":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD","download_json":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD.json","view_paper":"https://pith.science/paper/7XSJ2CHZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12031&json=true","fetch_graph":"https://pith.science/api/pith-number/7XSJ2CHZR7MOQVHQVBSWVZCMLD/graph.json","fetch_events":"https://pith.science/api/pith-number/7XSJ2CHZR7MOQVHQVBSWVZCMLD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD/action/storage_attestation","attest_author":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD/action/author_attestation","sign_citation":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD/action/citation_signature","submit_replication":"https://pith.science/pith/7XSJ2CHZR7MOQVHQVBSWVZCMLD/action/replication_record"}},"created_at":"2026-07-05T09:38:27.356179+00:00","updated_at":"2026-07-05T09:38:27.356179+00:00"}