{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ETNIKHN5RQSMF5R2PISNAAU54N","short_pith_number":"pith:ETNIKHN5","schema_version":"1.0","canonical_sha256":"24da851dbd8c24c2f63a7a24d0029de35206ca60f5ad47b4a3639112726311df","source":{"kind":"arxiv","id":"2303.07272","version":6},"attestation_state":"computed","paper":{"title":"Accounting for multiplicity in machine learning benchmark performance","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ME","authors_text":"Einar Holsb{\\o}, Kajsa M{\\o}llersen","submitted_at":"2023-03-10T10:32:18Z","abstract_excerpt":"State-of-the-art (SOTA) performance refers to the highest performance achieved by some model on a test sample, preferably under controlled conditions such as public data (reproducibility) or public challenges (independent sample). Thousands of classifiers are applied, and the highest performance becomes the new reference point for a particular problem. In effect, this set-up is an estimate of the expected best performance among all classifiers applied to a random sample; a sample maximum estimate. In this paper, we argue that SOTA should instead be estimated by the expected performance of the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.07272","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"stat.ME","submitted_at":"2023-03-10T10:32:18Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3fe52c6c155a974cbf0ae7f3ab48807a75c94f3c2d38f7fc3846d4dbad24f5b8","abstract_canon_sha256":"1a9098c668a5c746936697234fa14d66af5c6f63092d49bca62b381f4768aaf5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:53.846870Z","signature_b64":"FTucV3baM54CKQYGT5IHx9dtuEUOD/L0iYirVYhBp7xc7FBk1RN5VZY1Luv8RpjgFAI5Eq99zhzGPqQtGnRKBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24da851dbd8c24c2f63a7a24d0029de35206ca60f5ad47b4a3639112726311df","last_reissued_at":"2026-07-05T11:36:53.846308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:53.846308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Accounting for multiplicity in machine learning benchmark performance","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ME","authors_text":"Einar Holsb{\\o}, Kajsa M{\\o}llersen","submitted_at":"2023-03-10T10:32:18Z","abstract_excerpt":"State-of-the-art (SOTA) performance refers to the highest performance achieved by some model on a test sample, preferably under controlled conditions such as public data (reproducibility) or public challenges (independent sample). Thousands of classifiers are applied, and the highest performance becomes the new reference point for a particular problem. In effect, this set-up is an estimate of the expected best performance among all classifiers applied to a random sample; a sample maximum estimate. In this paper, we argue that SOTA should instead be estimated by the expected performance of the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.07272","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.07272/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.07272","created_at":"2026-07-05T11:36:53.846379+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.07272v6","created_at":"2026-07-05T11:36:53.846379+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.07272","created_at":"2026-07-05T11:36:53.846379+00:00"},{"alias_kind":"pith_short_12","alias_value":"ETNIKHN5RQSM","created_at":"2026-07-05T11:36:53.846379+00:00"},{"alias_kind":"pith_short_16","alias_value":"ETNIKHN5RQSMF5R2","created_at":"2026-07-05T11:36:53.846379+00:00"},{"alias_kind":"pith_short_8","alias_value":"ETNIKHN5","created_at":"2026-07-05T11:36:53.846379+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.14959","citing_title":"Systemizing Multiplicity: The Curious Case of Arbitrariness in Machine Learning","ref_index":2016,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N","json":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N.json","graph_json":"https://pith.science/api/pith-number/ETNIKHN5RQSMF5R2PISNAAU54N/graph.json","events_json":"https://pith.science/api/pith-number/ETNIKHN5RQSMF5R2PISNAAU54N/events.json","paper":"https://pith.science/paper/ETNIKHN5"},"agent_actions":{"view_html":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N","download_json":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N.json","view_paper":"https://pith.science/paper/ETNIKHN5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.07272&json=true","fetch_graph":"https://pith.science/api/pith-number/ETNIKHN5RQSMF5R2PISNAAU54N/graph.json","fetch_events":"https://pith.science/api/pith-number/ETNIKHN5RQSMF5R2PISNAAU54N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N/action/storage_attestation","attest_author":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N/action/author_attestation","sign_citation":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N/action/citation_signature","submit_replication":"https://pith.science/pith/ETNIKHN5RQSMF5R2PISNAAU54N/action/replication_record"}},"created_at":"2026-07-05T11:36:53.846379+00:00","updated_at":"2026-07-05T11:36:53.846379+00:00"}