{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NDFFWIO2QPESQTMCHRS6ML3RRU","short_pith_number":"pith:NDFFWIO2","schema_version":"1.0","canonical_sha256":"68ca5b21da83c9284d823c65e62f718d307db5b48a66df9be26be38eee90d35d","source":{"kind":"arxiv","id":"2504.16109","version":1},"attestation_state":"computed","paper":{"title":"Representation Learning for Tabular Data: A Comprehensive Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Han-Jia Ye, Hao-Run Cai, Jun-Peng Jiang, Qile Zhou, Si-Yang Liu","submitted_at":"2025-04-17T17:58:23Z","abstract_excerpt":"Tabular data, structured as rows and columns, is among the most prevalent data types in machine learning classification and regression applications. Models for learning from tabular data have continuously evolved, with Deep Neural Networks (DNNs) recently demonstrating promising results through their capability of representation learning. In this survey, we systematically introduce the field of tabular representation learning, covering the background, challenges, and benchmarks, along with the pros and cons of using DNNs. We organize existing methods into three main categories according to the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16109","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-17T17:58:23Z","cross_cats_sorted":[],"title_canon_sha256":"eb2f75d3ca39970eef5a705f260719800602abb08dcc3768af7769adcb2be8d4","abstract_canon_sha256":"3a8de24837593b3c589dafe71ccf39935b9103a6c31522d36fc676b2d8238b8c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:41.550115Z","signature_b64":"WtsiFq4rDZwTajDtJrJ+vx+hY8RaBHzVH+v22o58AQ7nFMwk17NiK5uxEyQtF430CP3y4HnbapTc/f3wmCwEDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68ca5b21da83c9284d823c65e62f718d307db5b48a66df9be26be38eee90d35d","last_reissued_at":"2026-07-05T10:52:41.549668Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:41.549668Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Representation Learning for Tabular Data: A Comprehensive Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Han-Jia Ye, Hao-Run Cai, Jun-Peng Jiang, Qile Zhou, Si-Yang Liu","submitted_at":"2025-04-17T17:58:23Z","abstract_excerpt":"Tabular data, structured as rows and columns, is among the most prevalent data types in machine learning classification and regression applications. Models for learning from tabular data have continuously evolved, with Deep Neural Networks (DNNs) recently demonstrating promising results through their capability of representation learning. In this survey, we systematically introduce the field of tabular representation learning, covering the background, challenges, and benchmarks, along with the pros and cons of using DNNs. We organize existing methods into three main categories according to the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16109","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16109","created_at":"2026-07-05T10:52:41.549724+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16109v1","created_at":"2026-07-05T10:52:41.549724+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16109","created_at":"2026-07-05T10:52:41.549724+00:00"},{"alias_kind":"pith_short_12","alias_value":"NDFFWIO2QPES","created_at":"2026-07-05T10:52:41.549724+00:00"},{"alias_kind":"pith_short_16","alias_value":"NDFFWIO2QPESQTMC","created_at":"2026-07-05T10:52:41.549724+00:00"},{"alias_kind":"pith_short_8","alias_value":"NDFFWIO2","created_at":"2026-07-05T10:52:41.549724+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07345","citing_title":"TabSwift: An Efficient Tabular Foundation Model with Row-Wise Attention","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30702","citing_title":"Accelerometry-Derived Digital Biomarkers for Cardiometabolic Risk: A Population-Representative Tabular Benchmark with Uncertainty Quantification","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18147","citing_title":"Foundation Models for Credit Risk Prediction: A Game Changer?","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02978","citing_title":"On the Robustness of Tabular Foundation Models: Test-Time Attacks and In-Context Defenses","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16791","citing_title":"TabArena: A Living Benchmark for Machine Learning on Tabular Data","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02519","citing_title":"Evaluating Tabular Representation Learning for Network Intrusion Detection","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU","json":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU.json","graph_json":"https://pith.science/api/pith-number/NDFFWIO2QPESQTMCHRS6ML3RRU/graph.json","events_json":"https://pith.science/api/pith-number/NDFFWIO2QPESQTMCHRS6ML3RRU/events.json","paper":"https://pith.science/paper/NDFFWIO2"},"agent_actions":{"view_html":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU","download_json":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU.json","view_paper":"https://pith.science/paper/NDFFWIO2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16109&json=true","fetch_graph":"https://pith.science/api/pith-number/NDFFWIO2QPESQTMCHRS6ML3RRU/graph.json","fetch_events":"https://pith.science/api/pith-number/NDFFWIO2QPESQTMCHRS6ML3RRU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU/action/storage_attestation","attest_author":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU/action/author_attestation","sign_citation":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU/action/citation_signature","submit_replication":"https://pith.science/pith/NDFFWIO2QPESQTMCHRS6ML3RRU/action/replication_record"}},"created_at":"2026-07-05T10:52:41.549724+00:00","updated_at":"2026-07-05T10:52:41.549724+00:00"}