{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:JL5RKDICUQBJAFXXHOL72SV7ID","short_pith_number":"pith:JL5RKDIC","schema_version":"1.0","canonical_sha256":"4afb150d02a4029016f73b97fd4abf40c1f04fe32f3e2fe2850490d743e54e37","source":{"kind":"arxiv","id":"2103.03098","version":1},"attestation_state":"computed","paper":{"title":"Accounting for Variance in Machine Learning Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Assya Trofimov, Brennan Nichyporuk, Chris Pal, Dmitriy Serdyuk, Edward Raff, Ga\\\"el Varoquaux, Justin Szeto, Kanika Madan, Mirko Bronzi, Naz Sepah, Pascal Vincent, Pierre Delaunay, Samira Ebrahimi Kahou, Tal Arbel, Vikram Voleti, Vincent Michalski, Xavier Bouthillier","submitted_at":"2021-03-01T22:39:49Z","abstract_excerpt":"Strong empirical evidence that one machine-learning algorithm A outperforms another one B ideally calls for multiple trials optimizing the learning pipeline over sources of variation such as data sampling, data augmentation, parameter initialization, and hyperparameters choices. This is prohibitively expensive, and corners are cut to reach conclusions. We model the whole benchmarking process, revealing that variance due to data sampling, parameter initialization and hyperparameter choice impact markedly the results. We analyze the predominant comparison methods used today in the light of this "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.03098","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-01T22:39:49Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"27a4645364d0de9f49b7663bb976996657450ce524f4a58528af07900c8427fd","abstract_canon_sha256":"9a3fa58118b1397f9df18aaf8f09bb4b2c24f65dd58941492f6e8c7f66514849"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:20:24.278733Z","signature_b64":"1r3OFRpu/JrJY1BP/2D4YgdjCcfaEUfuKjK5Ja1USlztBXPqSfuFeXLOTOxOWc38EJQ740lZGrd1P994ay2JCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4afb150d02a4029016f73b97fd4abf40c1f04fe32f3e2fe2850490d743e54e37","last_reissued_at":"2026-07-05T02:20:24.278242Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:20:24.278242Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Accounting for Variance in Machine Learning Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Assya Trofimov, Brennan Nichyporuk, Chris Pal, Dmitriy Serdyuk, Edward Raff, Ga\\\"el Varoquaux, Justin Szeto, Kanika Madan, Mirko Bronzi, Naz Sepah, Pascal Vincent, Pierre Delaunay, Samira Ebrahimi Kahou, Tal Arbel, Vikram Voleti, Vincent Michalski, Xavier Bouthillier","submitted_at":"2021-03-01T22:39:49Z","abstract_excerpt":"Strong empirical evidence that one machine-learning algorithm A outperforms another one B ideally calls for multiple trials optimizing the learning pipeline over sources of variation such as data sampling, data augmentation, parameter initialization, and hyperparameters choices. This is prohibitively expensive, and corners are cut to reach conclusions. We model the whole benchmarking process, revealing that variance due to data sampling, parameter initialization and hyperparameter choice impact markedly the results. We analyze the predominant comparison methods used today in the light of this "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.03098","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.03098/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.03098","created_at":"2026-07-05T02:20:24.278301+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.03098v1","created_at":"2026-07-05T02:20:24.278301+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.03098","created_at":"2026-07-05T02:20:24.278301+00:00"},{"alias_kind":"pith_short_12","alias_value":"JL5RKDICUQBJ","created_at":"2026-07-05T02:20:24.278301+00:00"},{"alias_kind":"pith_short_16","alias_value":"JL5RKDICUQBJAFXX","created_at":"2026-07-05T02:20:24.278301+00:00"},{"alias_kind":"pith_short_8","alias_value":"JL5RKDIC","created_at":"2026-07-05T02:20:24.278301+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24996","citing_title":"From Forecasting Leaderboards to Deployment Decisions: A Fail-Closed Certification Protocol","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05932","citing_title":"A Pre-Registered Causal Partition of Self-Consistency Elicitation and Reward Design in RLVR","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25492","citing_title":"SafetyRepro: Configuration-Conditional Rank Instability on Alignment Benchmarks","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17842","citing_title":"QuickScope: Certifying Hard Questions in Dynamic LLM Benchmarks","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID","json":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID.json","graph_json":"https://pith.science/api/pith-number/JL5RKDICUQBJAFXXHOL72SV7ID/graph.json","events_json":"https://pith.science/api/pith-number/JL5RKDICUQBJAFXXHOL72SV7ID/events.json","paper":"https://pith.science/paper/JL5RKDIC"},"agent_actions":{"view_html":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID","download_json":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID.json","view_paper":"https://pith.science/paper/JL5RKDIC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.03098&json=true","fetch_graph":"https://pith.science/api/pith-number/JL5RKDICUQBJAFXXHOL72SV7ID/graph.json","fetch_events":"https://pith.science/api/pith-number/JL5RKDICUQBJAFXXHOL72SV7ID/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID/action/storage_attestation","attest_author":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID/action/author_attestation","sign_citation":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID/action/citation_signature","submit_replication":"https://pith.science/pith/JL5RKDICUQBJAFXXHOL72SV7ID/action/replication_record"}},"created_at":"2026-07-05T02:20:24.278301+00:00","updated_at":"2026-07-05T02:20:24.278301+00:00"}