{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SSTBDB7NNVWBDHUMD6SVC63SZG","short_pith_number":"pith:SSTBDB7N","schema_version":"1.0","canonical_sha256":"94a61187ed6d6c119e8c1fa5517b72c9b5189d761c103d9f94794e625722fc5f","source":{"kind":"arxiv","id":"2311.02971","version":3},"attestation_state":"computed","paper":{"title":"TabRepo: A Large Scale Repository of Tabular Model Evaluations and its AutoML Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Salinas, Nick Erickson","submitted_at":"2023-11-06T09:17:18Z","abstract_excerpt":"We introduce TabRepo, a new dataset of tabular model evaluations and predictions. TabRepo contains the predictions and metrics of 1310 models evaluated on 200 classification and regression datasets. We illustrate the benefit of our dataset in multiple ways. First, we show that it allows to perform analysis such as comparing Hyperparameter Optimization against current AutoML systems while also considering ensembling at marginal cost by using precomputed model predictions. Second, we show that our dataset can be readily leveraged to perform transfer-learning. In particular, we show that applying"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.02971","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-06T09:17:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1f064bb12089012da788f4a0720c29f3878ded95476e7fb96e6d7f1238b7a208","abstract_canon_sha256":"da066367f4b06b88f3b546c53b1b41f0b5f0d54a0a55ff0396afff65c36b92bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:04.076332Z","signature_b64":"B5SNTge9axHSG+KN7DPusv15qzY/eWgMVUWBbhKhADoJ01DO/n+5hG5pz8lC66BO+DHy4L2zci8HX7bB+yH5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"94a61187ed6d6c119e8c1fa5517b72c9b5189d761c103d9f94794e625722fc5f","last_reissued_at":"2026-07-05T08:59:04.075821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:04.075821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TabRepo: A Large Scale Repository of Tabular Model Evaluations and its AutoML Applications","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Salinas, Nick Erickson","submitted_at":"2023-11-06T09:17:18Z","abstract_excerpt":"We introduce TabRepo, a new dataset of tabular model evaluations and predictions. TabRepo contains the predictions and metrics of 1310 models evaluated on 200 classification and regression datasets. We illustrate the benefit of our dataset in multiple ways. First, we show that it allows to perform analysis such as comparing Hyperparameter Optimization against current AutoML systems while also considering ensembling at marginal cost by using precomputed model predictions. Second, we show that our dataset can be readily leveraged to perform transfer-learning. In particular, we show that applying"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02971","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02971/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.02971","created_at":"2026-07-05T08:59:04.075876+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.02971v3","created_at":"2026-07-05T08:59:04.075876+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02971","created_at":"2026-07-05T08:59:04.075876+00:00"},{"alias_kind":"pith_short_12","alias_value":"SSTBDB7NNVWB","created_at":"2026-07-05T08:59:04.075876+00:00"},{"alias_kind":"pith_short_16","alias_value":"SSTBDB7NNVWBDHUM","created_at":"2026-07-05T08:59:04.075876+00:00"},{"alias_kind":"pith_short_8","alias_value":"SSTBDB7N","created_at":"2026-07-05T08:59:04.075876+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05139","citing_title":"BBOmix: A Tabular Benchmark for Hyperparameter Optimization of Unsupervised Biological Representation Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23417","citing_title":"An Open-Source Training Dataset for Foundation Models for Black-box Optimization","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG","json":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG.json","graph_json":"https://pith.science/api/pith-number/SSTBDB7NNVWBDHUMD6SVC63SZG/graph.json","events_json":"https://pith.science/api/pith-number/SSTBDB7NNVWBDHUMD6SVC63SZG/events.json","paper":"https://pith.science/paper/SSTBDB7N"},"agent_actions":{"view_html":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG","download_json":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG.json","view_paper":"https://pith.science/paper/SSTBDB7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.02971&json=true","fetch_graph":"https://pith.science/api/pith-number/SSTBDB7NNVWBDHUMD6SVC63SZG/graph.json","fetch_events":"https://pith.science/api/pith-number/SSTBDB7NNVWBDHUMD6SVC63SZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG/action/storage_attestation","attest_author":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG/action/author_attestation","sign_citation":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG/action/citation_signature","submit_replication":"https://pith.science/pith/SSTBDB7NNVWBDHUMD6SVC63SZG/action/replication_record"}},"created_at":"2026-07-05T08:59:04.075876+00:00","updated_at":"2026-07-05T08:59:04.075876+00:00"}