{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5F3DW7VQBJTMO5I66SR6OWM3SY","short_pith_number":"pith:5F3DW7VQ","schema_version":"1.0","canonical_sha256":"e9763b7eb00a66c7751ef4a3e7599b960639bd0650e74ee95a3c8cb36d591450","source":{"kind":"arxiv","id":"2106.11959","version":5},"attestation_state":"computed","paper":{"title":"Revisiting Deep Learning Models for Tabular Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Artem Babenko, Ivan Rubachev, Valentin Khrulkov, Yury Gorishniy","submitted_at":"2021-06-22T17:58:10Z","abstract_excerpt":"The existing literature on deep learning for tabular data proposes a wide range of novel architectures and reports competitive results on various datasets. However, the proposed models are usually not properly compared to each other and existing works often use different benchmarks and experiment protocols. As a result, it is unclear for both researchers and practitioners what models perform best. Additionally, the field still lacks effective baselines, that is, the easy-to-use models that provide competitive performance across different problems.\n  In this work, we perform an overview of the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.11959","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-22T17:58:10Z","cross_cats_sorted":[],"title_canon_sha256":"1ba3bbe62b5603528d2932137df2b47265326c91e0068b5ce4236073db263be8","abstract_canon_sha256":"a9ceadfc7e17fc96b363d4a2e2069fe631ab3083fe72ae4c6c8d464e3ef70e2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:03.741499Z","signature_b64":"hEzQTz1o++HQrOOcQp21UBf2FFt+NKcCOoQlhe8UHkfG2oH68ZgM/NHzuvdVTId85rl4nGJkXC4feq8k8JkbCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e9763b7eb00a66c7751ef4a3e7599b960639bd0650e74ee95a3c8cb36d591450","last_reissued_at":"2026-07-05T07:05:03.740924Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:03.740924Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Deep Learning Models for Tabular Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Artem Babenko, Ivan Rubachev, Valentin Khrulkov, Yury Gorishniy","submitted_at":"2021-06-22T17:58:10Z","abstract_excerpt":"The existing literature on deep learning for tabular data proposes a wide range of novel architectures and reports competitive results on various datasets. However, the proposed models are usually not properly compared to each other and existing works often use different benchmarks and experiment protocols. As a result, it is unclear for both researchers and practitioners what models perform best. Additionally, the field still lacks effective baselines, that is, the easy-to-use models that provide competitive performance across different problems.\n  In this work, we perform an overview of the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.11959","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.11959/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.11959","created_at":"2026-07-05T07:05:03.740985+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.11959v5","created_at":"2026-07-05T07:05:03.740985+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.11959","created_at":"2026-07-05T07:05:03.740985+00:00"},{"alias_kind":"pith_short_12","alias_value":"5F3DW7VQBJTM","created_at":"2026-07-05T07:05:03.740985+00:00"},{"alias_kind":"pith_short_16","alias_value":"5F3DW7VQBJTMO5I6","created_at":"2026-07-05T07:05:03.740985+00:00"},{"alias_kind":"pith_short_8","alias_value":"5F3DW7VQ","created_at":"2026-07-05T07:05:03.740985+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.21876","citing_title":"Comparative Evaluation of Machine Learning Models for Predicting Donor Kidney Discard","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16378","citing_title":"Reciprocal Co-Training (RCT): Coupling Gradient-Based and Non-Differentiable Models via Reinforcement Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25154","citing_title":"Prior-Aligned Data Cleaning for Tabular Foundation Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05257","citing_title":"Extending Tabular Denoising Diffusion Probabilistic Models for Time-Series Data Generation","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY","json":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY.json","graph_json":"https://pith.science/api/pith-number/5F3DW7VQBJTMO5I66SR6OWM3SY/graph.json","events_json":"https://pith.science/api/pith-number/5F3DW7VQBJTMO5I66SR6OWM3SY/events.json","paper":"https://pith.science/paper/5F3DW7VQ"},"agent_actions":{"view_html":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY","download_json":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY.json","view_paper":"https://pith.science/paper/5F3DW7VQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.11959&json=true","fetch_graph":"https://pith.science/api/pith-number/5F3DW7VQBJTMO5I66SR6OWM3SY/graph.json","fetch_events":"https://pith.science/api/pith-number/5F3DW7VQBJTMO5I66SR6OWM3SY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY/action/storage_attestation","attest_author":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY/action/author_attestation","sign_citation":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY/action/citation_signature","submit_replication":"https://pith.science/pith/5F3DW7VQBJTMO5I66SR6OWM3SY/action/replication_record"}},"created_at":"2026-07-05T07:05:03.740985+00:00","updated_at":"2026-07-05T07:05:03.740985+00:00"}