{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TWP2EUVREJXII4CP2KYLSJRSZ7","short_pith_number":"pith:TWP2EUVR","schema_version":"1.0","canonical_sha256":"9d9fa252b1226e84704fd2b0b92632cfc4403070a4b7df753ff1ae0435e525da","source":{"kind":"arxiv","id":"2505.14415","version":2},"attestation_state":"computed","paper":{"title":"Table Foundation Models: on knowledge pre-training for tabular learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexandre Perez-Lebel, F\\'elix Lefebvre, Ga\\\"el Varoquaux, Ga\\\"etan Brison, Myung Jun Kim","submitted_at":"2025-05-20T14:27:51Z","abstract_excerpt":"Table foundation models bring high hopes to data science: pre-trained on tabular data to embark knowledge or priors, they should facilitate downstream tasks on tables. One specific challenge is that of data semantics: numerical entries take their meaning from context, e.g., column name. Pre-trained neural networks that jointly model column names and table entries have recently boosted prediction accuracy. While these models outline the promises of world knowledge to interpret table values, they lack the convenience of popular foundation models in text or vision. Indeed, they must be fine-tuned"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14415","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-20T14:27:51Z","cross_cats_sorted":[],"title_canon_sha256":"8e2c6b8a3587e8c9909f5370dbdc8cc519fb2688bb5952b0607758ca6926b2d2","abstract_canon_sha256":"80349d3c58dd2a0c8ffdeecc1d7b18a7b3fba0c04614b3ecba71897e35697dd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:11.719726Z","signature_b64":"PYQNd6SLxvsMq1NYokcUUhhWVylgAi0818gO0656p4QMK95TLafmtIDvEVHDsZ1tJ75yZdBWB9riFh7fKKF8Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d9fa252b1226e84704fd2b0b92632cfc4403070a4b7df753ff1ae0435e525da","last_reissued_at":"2026-07-05T11:29:11.719135Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:11.719135Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Table Foundation Models: on knowledge pre-training for tabular learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexandre Perez-Lebel, F\\'elix Lefebvre, Ga\\\"el Varoquaux, Ga\\\"etan Brison, Myung Jun Kim","submitted_at":"2025-05-20T14:27:51Z","abstract_excerpt":"Table foundation models bring high hopes to data science: pre-trained on tabular data to embark knowledge or priors, they should facilitate downstream tasks on tables. One specific challenge is that of data semantics: numerical entries take their meaning from context, e.g., column name. Pre-trained neural networks that jointly model column names and table entries have recently boosted prediction accuracy. While these models outline the promises of world knowledge to interpret table values, they lack the convenience of popular foundation models in text or vision. Indeed, they must be fine-tuned"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14415","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14415/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14415","created_at":"2026-07-05T11:29:11.719209+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14415v2","created_at":"2026-07-05T11:29:11.719209+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14415","created_at":"2026-07-05T11:29:11.719209+00:00"},{"alias_kind":"pith_short_12","alias_value":"TWP2EUVREJXI","created_at":"2026-07-05T11:29:11.719209+00:00"},{"alias_kind":"pith_short_16","alias_value":"TWP2EUVREJXII4CP","created_at":"2026-07-05T11:29:11.719209+00:00"},{"alias_kind":"pith_short_8","alias_value":"TWP2EUVR","created_at":"2026-07-05T11:29:11.719209+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30410","citing_title":"Beyond IID: How General Are Tabular Foundation Models, Really?","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10616","citing_title":"MulTaBench: Benchmarking Multimodal Tabular Learning with Text and Image","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02003","citing_title":"RamanBench: A Large-Scale Benchmark for Machine Learning on Raman Spectroscopy","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7","json":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7.json","graph_json":"https://pith.science/api/pith-number/TWP2EUVREJXII4CP2KYLSJRSZ7/graph.json","events_json":"https://pith.science/api/pith-number/TWP2EUVREJXII4CP2KYLSJRSZ7/events.json","paper":"https://pith.science/paper/TWP2EUVR"},"agent_actions":{"view_html":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7","download_json":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7.json","view_paper":"https://pith.science/paper/TWP2EUVR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14415&json=true","fetch_graph":"https://pith.science/api/pith-number/TWP2EUVREJXII4CP2KYLSJRSZ7/graph.json","fetch_events":"https://pith.science/api/pith-number/TWP2EUVREJXII4CP2KYLSJRSZ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7/action/storage_attestation","attest_author":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7/action/author_attestation","sign_citation":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7/action/citation_signature","submit_replication":"https://pith.science/pith/TWP2EUVREJXII4CP2KYLSJRSZ7/action/replication_record"}},"created_at":"2026-07-05T11:29:11.719209+00:00","updated_at":"2026-07-05T11:29:11.719209+00:00"}