{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5GSQ2YXCZDMKFZYRHW57WUQ2F4","short_pith_number":"pith:5GSQ2YXC","schema_version":"1.0","canonical_sha256":"e9a50d62e2c8d8a2e7113dbbfb521a2f19c85697b622b4951a6bdbfc59bb444e","source":{"kind":"arxiv","id":"2111.02705","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking Multimodal AutoML for Tabular Data with Text Fields","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander J. Smola, Jonas Mueller, Mu Li, Nick Erickson, Xingjian Shi","submitted_at":"2021-11-04T09:29:16Z","abstract_excerpt":"We consider the use of automated supervised learning systems for data tables that not only contain numeric/categorical columns, but one or more text fields as well. Here we assemble 18 multimodal data tables that each contain some text fields and stem from a real business application. Our publicly-available benchmark enables researchers to comprehensively evaluate their own methods for supervised learning with numeric, categorical, and text features. To ensure that any single modeling strategy which performs well over all 18 datasets will serve as a practical foundation for multimodal text/tab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.02705","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-11-04T09:29:16Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"f8520a4d231c8ed3c9864c509ced234d5c8502aef1e41731cbe62d9178061bfe","abstract_canon_sha256":"ec3c5fdcaf8947af0c5df3c9c219224c217a77dc9a8b17b778823dffe97dc37c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:29:07.604926Z","signature_b64":"KKQKXmuHHO5mbf2VFMOngsAAQ7vGtlsoGwGOvwVYIC11fWO0py45TDqJyftm1kvQtWfQOmIPFKG61JmnmDjvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e9a50d62e2c8d8a2e7113dbbfb521a2f19c85697b622b4951a6bdbfc59bb444e","last_reissued_at":"2026-07-05T03:29:07.604371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:29:07.604371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Multimodal AutoML for Tabular Data with Text Fields","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander J. Smola, Jonas Mueller, Mu Li, Nick Erickson, Xingjian Shi","submitted_at":"2021-11-04T09:29:16Z","abstract_excerpt":"We consider the use of automated supervised learning systems for data tables that not only contain numeric/categorical columns, but one or more text fields as well. Here we assemble 18 multimodal data tables that each contain some text fields and stem from a real business application. Our publicly-available benchmark enables researchers to comprehensively evaluate their own methods for supervised learning with numeric, categorical, and text features. To ensure that any single modeling strategy which performs well over all 18 datasets will serve as a practical foundation for multimodal text/tab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.02705","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.02705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.02705","created_at":"2026-07-05T03:29:07.604440+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.02705v1","created_at":"2026-07-05T03:29:07.604440+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.02705","created_at":"2026-07-05T03:29:07.604440+00:00"},{"alias_kind":"pith_short_12","alias_value":"5GSQ2YXCZDMK","created_at":"2026-07-05T03:29:07.604440+00:00"},{"alias_kind":"pith_short_16","alias_value":"5GSQ2YXCZDMKFZYR","created_at":"2026-07-05T03:29:07.604440+00:00"},{"alias_kind":"pith_short_8","alias_value":"5GSQ2YXC","created_at":"2026-07-05T03:29:07.604440+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13986","citing_title":"TabPFN-3: Technical Report","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06806","citing_title":"MachineLearningLM: Scaling Many-shot In-context Learning via Continued Pretraining","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13986","citing_title":"TabPFN-3: Technical Report","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12292","citing_title":"STRABLE: Benchmarking Tabular Machine Learning with Strings","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4","json":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4.json","graph_json":"https://pith.science/api/pith-number/5GSQ2YXCZDMKFZYRHW57WUQ2F4/graph.json","events_json":"https://pith.science/api/pith-number/5GSQ2YXCZDMKFZYRHW57WUQ2F4/events.json","paper":"https://pith.science/paper/5GSQ2YXC"},"agent_actions":{"view_html":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4","download_json":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4.json","view_paper":"https://pith.science/paper/5GSQ2YXC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.02705&json=true","fetch_graph":"https://pith.science/api/pith-number/5GSQ2YXCZDMKFZYRHW57WUQ2F4/graph.json","fetch_events":"https://pith.science/api/pith-number/5GSQ2YXCZDMKFZYRHW57WUQ2F4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4/action/storage_attestation","attest_author":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4/action/author_attestation","sign_citation":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4/action/citation_signature","submit_replication":"https://pith.science/pith/5GSQ2YXCZDMKFZYRHW57WUQ2F4/action/replication_record"}},"created_at":"2026-07-05T03:29:07.604440+00:00","updated_at":"2026-07-05T03:29:07.604440+00:00"}