{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:HE5DMRNNTU6QWZC6YEW4MFSYPR","short_pith_number":"pith:HE5DMRNN","schema_version":"1.0","canonical_sha256":"393a3645ad9d3d0b645ec12dc616587c67278e390498d8afc711037687a72e47","source":{"kind":"arxiv","id":"1910.05446","version":3},"attestation_state":"computed","paper":{"title":"On Empirical Comparisons of Optimizers for Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Chris J. Maddison, Christopher J. Shallue, Dami Choi, George E. Dahl, Jaehoon Lee, Zachary Nado","submitted_at":"2019-10-11T23:51:09Z","abstract_excerpt":"Selecting an optimizer is a central step in the contemporary deep learning pipeline. In this paper, we demonstrate the sensitivity of optimizer comparisons to the hyperparameter tuning protocol. Our findings suggest that the hyperparameter search space may be the single most important factor explaining the rankings obtained by recent empirical comparisons in the literature. In fact, we show that these results can be contradicted when hyperparameter search spaces are changed. As tuning effort grows without bound, more general optimizers should never underperform the ones they can approximate (i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.05446","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-11T23:51:09Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"6dc1dc29e268dccad636d37a03a79b4484d4d73b256cf0634b8c72d2502fb3c2","abstract_canon_sha256":"7c67d1270aef30b00a77d52e1880e9f9580497b51cb64d01c59338e3aa899232"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:10:28.924672Z","signature_b64":"nSoZh6c408/q17xoJgN8sqFkHFJY7eNqnQZnYL+Mfoi20BhMji/Vuu4/+h1xoRF94EgmB4sZB0++iFUvbYhoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"393a3645ad9d3d0b645ec12dc616587c67278e390498d8afc711037687a72e47","last_reissued_at":"2026-07-05T01:10:28.924183Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:10:28.924183Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Empirical Comparisons of Optimizers for Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Chris J. Maddison, Christopher J. Shallue, Dami Choi, George E. Dahl, Jaehoon Lee, Zachary Nado","submitted_at":"2019-10-11T23:51:09Z","abstract_excerpt":"Selecting an optimizer is a central step in the contemporary deep learning pipeline. In this paper, we demonstrate the sensitivity of optimizer comparisons to the hyperparameter tuning protocol. Our findings suggest that the hyperparameter search space may be the single most important factor explaining the rankings obtained by recent empirical comparisons in the literature. In fact, we show that these results can be contradicted when hyperparameter search spaces are changed. As tuning effort grows without bound, more general optimizers should never underperform the ones they can approximate (i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.05446","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.05446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.05446","created_at":"2026-07-05T01:10:28.924242+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.05446v3","created_at":"2026-07-05T01:10:28.924242+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.05446","created_at":"2026-07-05T01:10:28.924242+00:00"},{"alias_kind":"pith_short_12","alias_value":"HE5DMRNNTU6Q","created_at":"2026-07-05T01:10:28.924242+00:00"},{"alias_kind":"pith_short_16","alias_value":"HE5DMRNNTU6QWZC6","created_at":"2026-07-05T01:10:28.924242+00:00"},{"alias_kind":"pith_short_8","alias_value":"HE5DMRNN","created_at":"2026-07-05T01:10:28.924242+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22644","citing_title":"Why SGD is not Brownian Motion: A New Perspective on Stochastic Dynamics","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15297","citing_title":"Benchmarking Optimizers for MLPs in Tabular Deep Learning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR","json":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR.json","graph_json":"https://pith.science/api/pith-number/HE5DMRNNTU6QWZC6YEW4MFSYPR/graph.json","events_json":"https://pith.science/api/pith-number/HE5DMRNNTU6QWZC6YEW4MFSYPR/events.json","paper":"https://pith.science/paper/HE5DMRNN"},"agent_actions":{"view_html":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR","download_json":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR.json","view_paper":"https://pith.science/paper/HE5DMRNN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.05446&json=true","fetch_graph":"https://pith.science/api/pith-number/HE5DMRNNTU6QWZC6YEW4MFSYPR/graph.json","fetch_events":"https://pith.science/api/pith-number/HE5DMRNNTU6QWZC6YEW4MFSYPR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR/action/storage_attestation","attest_author":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR/action/author_attestation","sign_citation":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR/action/citation_signature","submit_replication":"https://pith.science/pith/HE5DMRNNTU6QWZC6YEW4MFSYPR/action/replication_record"}},"created_at":"2026-07-05T01:10:28.924242+00:00","updated_at":"2026-07-05T01:10:28.924242+00:00"}