{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OSFM3QZF44KIRVBWOXOTG2W7KC","short_pith_number":"pith:OSFM3QZF","schema_version":"1.0","canonical_sha256":"748acdc325e71488d43675dd336adf509a7e31b87ec3d0701d828e7aa1c4f873","source":{"kind":"arxiv","id":"2405.01147","version":2},"attestation_state":"computed","paper":{"title":"Why Tabular Foundation Models Should Be a Research Priority","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Boris van Breugel, Mihaela van der Schaar","submitted_at":"2024-05-02T10:05:16Z","abstract_excerpt":"Recent text and image foundation models are incredibly impressive, and these models are attracting an ever-increasing portion of research resources. In this position piece we aim to shift the ML research community's priorities ever so slightly to a different modality: tabular data. Tabular data is the dominant modality in many fields, yet it is given hardly any research attention and significantly lags behind in terms of scale and power. We believe the time is now to start developing tabular foundation models, or what we coin a Large Tabular Model (LTM). LTMs could revolutionise the way scienc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.01147","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T10:05:16Z","cross_cats_sorted":[],"title_canon_sha256":"a006b05f087217cd0c6c72222ca7f251e2e91f6799a62a57a1919b2add8fa531","abstract_canon_sha256":"433eda46db52e2dc5e31f99a445d00e72d3da631a9d74f503ebe8b26a7811354"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:24.137987Z","signature_b64":"3xLHi7Fv0AcuxXXBkRyXSmzX+L+9YSVszJqyY9lklw2qmNlyK4L9jlxmoY5BF4T/bUJRZbiSyx7r99kgpKxpCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"748acdc325e71488d43675dd336adf509a7e31b87ec3d0701d828e7aa1c4f873","last_reissued_at":"2026-07-05T08:26:24.137514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:24.137514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Tabular Foundation Models Should Be a Research Priority","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Boris van Breugel, Mihaela van der Schaar","submitted_at":"2024-05-02T10:05:16Z","abstract_excerpt":"Recent text and image foundation models are incredibly impressive, and these models are attracting an ever-increasing portion of research resources. In this position piece we aim to shift the ML research community's priorities ever so slightly to a different modality: tabular data. Tabular data is the dominant modality in many fields, yet it is given hardly any research attention and significantly lags behind in terms of scale and power. We believe the time is now to start developing tabular foundation models, or what we coin a Large Tabular Model (LTM). LTMs could revolutionise the way scienc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01147","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.01147/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.01147","created_at":"2026-07-05T08:26:24.137572+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.01147v2","created_at":"2026-07-05T08:26:24.137572+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01147","created_at":"2026-07-05T08:26:24.137572+00:00"},{"alias_kind":"pith_short_12","alias_value":"OSFM3QZF44KI","created_at":"2026-07-05T08:26:24.137572+00:00"},{"alias_kind":"pith_short_16","alias_value":"OSFM3QZF44KIRVBW","created_at":"2026-07-05T08:26:24.137572+00:00"},{"alias_kind":"pith_short_8","alias_value":"OSFM3QZF","created_at":"2026-07-05T08:26:24.137572+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11473","citing_title":"CRUMB: Efficient Prior Fitted Network Inference via Distributionally Matched Context Batching","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31208","citing_title":"Probing Memorization of Tabular In-Context Learning","ref_index":186,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24680","citing_title":"Trajectory-Based Difficulty Scoring for Reliable Learning on Tabular Data","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30452","citing_title":"Exploring Differences Between Tabular Enterprise Data and Public Benchmarks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2501.01793","citing_title":"Creating Artificial Students that Never Existed: Leveraging Large Language Models and CTGANs for Synthetic Data Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02161","citing_title":"LLM-TabLogic: Preserving Inter-Column Logical Relationships in Synthetic Tabular Data via Prompt-Guided Latent Diffusion","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14998","citing_title":"Tables Guide Vision: Learning to See the Heart through Tabular Data","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19693","citing_title":"TREASURE: The Visa Payment Foundation Model for High-Volume Transaction Understanding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04292","citing_title":"SQuARE: Structured Query & Adaptive Retrieval Engine For Tabular Formats","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06047","citing_title":"TFM-Retouche: A Lightweight Input-Space Adapter for Tabular Foundation Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06290","citing_title":"Data Language Models: A New Foundation Model Class for Tabular Data","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06047","citing_title":"TFM-Retouche: A Lightweight Input-Space Adapter for Tabular Foundation Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18966","citing_title":"Self-Improving Tabular Language Models via Iterative Reward-Guided Post-Training","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04868","citing_title":"Noise Immunity in In-Context Tabular Learning: An Empirical Robustness Analysis of TabPFN's Attention Mechanisms","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC","json":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC.json","graph_json":"https://pith.science/api/pith-number/OSFM3QZF44KIRVBWOXOTG2W7KC/graph.json","events_json":"https://pith.science/api/pith-number/OSFM3QZF44KIRVBWOXOTG2W7KC/events.json","paper":"https://pith.science/paper/OSFM3QZF"},"agent_actions":{"view_html":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC","download_json":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC.json","view_paper":"https://pith.science/paper/OSFM3QZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.01147&json=true","fetch_graph":"https://pith.science/api/pith-number/OSFM3QZF44KIRVBWOXOTG2W7KC/graph.json","fetch_events":"https://pith.science/api/pith-number/OSFM3QZF44KIRVBWOXOTG2W7KC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC/action/storage_attestation","attest_author":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC/action/author_attestation","sign_citation":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC/action/citation_signature","submit_replication":"https://pith.science/pith/OSFM3QZF44KIRVBWOXOTG2W7KC/action/replication_record"}},"created_at":"2026-07-05T08:26:24.137572+00:00","updated_at":"2026-07-05T08:26:24.137572+00:00"}