{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:ZA3EPOEVBDT7I3P3NTY7ZAKUD7","short_pith_number":"pith:ZA3EPOEV","schema_version":"1.0","canonical_sha256":"c83647b89508e7f46dfb6cf1fc81541ff312df8954a9646a4276e509665fc1fa","source":{"kind":"arxiv","id":"1708.03731","version":3},"attestation_state":"computed","paper":{"title":"OpenML Benchmarking Suites","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Bernd Bischl, Frank Hutter, Giuseppe Casalicchio, Jan N. van Rijn, Joaquin Vanschoren, Matthias Feurer, Michel Lang, Pieter Gijsbers, Rafael G. Mantovani","submitted_at":"2017-08-11T23:28:48Z","abstract_excerpt":"Machine learning research depends on objectively interpretable, comparable, and reproducible algorithm benchmarks. We advocate the use of curated, comprehensive suites of machine learning tasks to standardize the setup, execution, and reporting of benchmarks. We enable this through software tools that help to create and leverage these benchmarking suites. These are seamlessly integrated into the OpenML platform, and accessible through interfaces in Python, Java, and R. OpenML benchmarking suites (a) are easy to use through standardized data formats, APIs, and client libraries; (b) come with ex"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1708.03731","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2017-08-11T23:28:48Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"af351827b0912ed4fcc8b1cad7d92cd83058713eaab1b757b076ad8db731d69f","abstract_canon_sha256":"18cef427f14bc746fb7b9f2783d8d05a30425b6d93ffe0d65e57dd491c1c927c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:10:18.522325Z","signature_b64":"uuXoGOuJ0jfAPvAvjhvgG+dWMsf1IcmPuQvsCpyYUvqXnTtANlaP7+SQNA43e/WXN0gezyVXNB6FkRpFx588Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c83647b89508e7f46dfb6cf1fc81541ff312df8954a9646a4276e509665fc1fa","last_reissued_at":"2026-07-05T07:10:18.521847Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:10:18.521847Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenML Benchmarking Suites","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Bernd Bischl, Frank Hutter, Giuseppe Casalicchio, Jan N. van Rijn, Joaquin Vanschoren, Matthias Feurer, Michel Lang, Pieter Gijsbers, Rafael G. Mantovani","submitted_at":"2017-08-11T23:28:48Z","abstract_excerpt":"Machine learning research depends on objectively interpretable, comparable, and reproducible algorithm benchmarks. We advocate the use of curated, comprehensive suites of machine learning tasks to standardize the setup, execution, and reporting of benchmarks. We enable this through software tools that help to create and leverage these benchmarking suites. These are seamlessly integrated into the OpenML platform, and accessible through interfaces in Python, Java, and R. OpenML benchmarking suites (a) are easy to use through standardized data formats, APIs, and client libraries; (b) come with ex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1708.03731","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1708.03731/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1708.03731","created_at":"2026-07-05T07:10:18.521906+00:00"},{"alias_kind":"arxiv_version","alias_value":"1708.03731v3","created_at":"2026-07-05T07:10:18.521906+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1708.03731","created_at":"2026-07-05T07:10:18.521906+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZA3EPOEVBDT7","created_at":"2026-07-05T07:10:18.521906+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZA3EPOEVBDT7I3P3","created_at":"2026-07-05T07:10:18.521906+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZA3EPOEV","created_at":"2026-07-05T07:10:18.521906+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04485","citing_title":"LimiX-2M: Mitigating Low-Rank Collapse and Attention Bottlenecks in Tabular Foundation Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02384","citing_title":"TabPrep: Closing the Feature Engineering Gap in Tabular Benchmarks","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04363","citing_title":"Mitigating Label Shift in Tabular In-Context Learning via Test-Time Posterior Adjustment","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22740","citing_title":"Ternary Decision Trees with Locally-Adaptive Uncertainty Zones","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30410","citing_title":"Beyond IID: How General Are Tabular Foundation Models, Really?","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"1907.00678","citing_title":"Two-stage Optimization for Machine Learning Workflow","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22740","citing_title":"Ternary Decision Trees with Locally-Adaptive Uncertainty Zones","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18971","citing_title":"Shaping the Prior: How Synthetic Task Distributions Determine Tabular Foundation Model Quality","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16791","citing_title":"TabArena: A Living Benchmark for Machine Learning on Tabular Data","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09329","citing_title":"MacrOData: New Benchmarks of Thousands of Datasets for Tabular Outlier Detection","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10315","citing_title":"Active Tabular Augmentation via Policy-Guided Diffusion Inpainting","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25154","citing_title":"Prior-Aligned Data Cleaning for Tabular Foundation Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06290","citing_title":"Data Language Models: A New Foundation Model Class for Tabular Data","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04962","citing_title":"TabEmbed: Benchmarking and Learning Generalist Embeddings for Tabular Understanding","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04363","citing_title":"Mitigating Label Shift in Tabular In-Context Learning via Test-Time Posterior Adjustment","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7","json":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7.json","graph_json":"https://pith.science/api/pith-number/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/graph.json","events_json":"https://pith.science/api/pith-number/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/events.json","paper":"https://pith.science/paper/ZA3EPOEV"},"agent_actions":{"view_html":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7","download_json":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7.json","view_paper":"https://pith.science/paper/ZA3EPOEV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1708.03731&json=true","fetch_graph":"https://pith.science/api/pith-number/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/graph.json","fetch_events":"https://pith.science/api/pith-number/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/action/storage_attestation","attest_author":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/action/author_attestation","sign_citation":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/action/citation_signature","submit_replication":"https://pith.science/pith/ZA3EPOEVBDT7I3P3NTY7ZAKUD7/action/replication_record"}},"created_at":"2026-07-05T07:10:18.521906+00:00","updated_at":"2026-07-05T07:10:18.521906+00:00"}