{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:YPDCWFK5KQ6LRFUCXY4O7BFD7U","short_pith_number":"pith:YPDCWFK5","schema_version":"1.0","canonical_sha256":"c3c62b155d543cb89682be38ef84a3fd3ca48c87ff89c032023aa0e2dd1c96e6","source":{"kind":"arxiv","id":"2006.15081","version":1},"attestation_state":"computed","paper":{"title":"On the Generalization Benefit of Noise in Stochastic Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Erich Elsen, Samuel L. Smith, Soham De","submitted_at":"2020-06-26T16:18:54Z","abstract_excerpt":"It has long been argued that minibatch stochastic gradient descent can generalize better than large batch gradient descent in deep neural networks. However recent papers have questioned this claim, arguing that this effect is simply a consequence of suboptimal hyperparameter tuning or insufficient compute budgets when the batch size is large. In this paper, we perform carefully designed experiments and rigorous hyperparameter sweeps on a range of popular models, which verify that small or moderately large batch sizes can substantially outperform very large batches on the test set. This occurs "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.15081","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-26T16:18:54Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"f97860d85c3a60405d45ecd9abe29c5a845d7729fb21134509d614831ef21337","abstract_canon_sha256":"4a1ca87f57962fe0485dcd7ba6c33e137ef6b22c774e6d224bb6fd585a0371b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:13:49.463276Z","signature_b64":"2NnVE7oVxrx6efFaSAJqSCzBiGTo8hALBfqYTxlW8d7UOqw0sVys0hz1kFMNY30/FEmoGJQwCDwos692mXLiCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3c62b155d543cb89682be38ef84a3fd3ca48c87ff89c032023aa0e2dd1c96e6","last_reissued_at":"2026-07-05T01:13:49.462869Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:13:49.462869Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Generalization Benefit of Noise in Stochastic Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Erich Elsen, Samuel L. Smith, Soham De","submitted_at":"2020-06-26T16:18:54Z","abstract_excerpt":"It has long been argued that minibatch stochastic gradient descent can generalize better than large batch gradient descent in deep neural networks. However recent papers have questioned this claim, arguing that this effect is simply a consequence of suboptimal hyperparameter tuning or insufficient compute budgets when the batch size is large. In this paper, we perform carefully designed experiments and rigorous hyperparameter sweeps on a range of popular models, which verify that small or moderately large batch sizes can substantially outperform very large batches on the test set. This occurs "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.15081","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.15081/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.15081","created_at":"2026-07-05T01:13:49.462933+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.15081v1","created_at":"2026-07-05T01:13:49.462933+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.15081","created_at":"2026-07-05T01:13:49.462933+00:00"},{"alias_kind":"pith_short_12","alias_value":"YPDCWFK5KQ6L","created_at":"2026-07-05T01:13:49.462933+00:00"},{"alias_kind":"pith_short_16","alias_value":"YPDCWFK5KQ6LRFUC","created_at":"2026-07-05T01:13:49.462933+00:00"},{"alias_kind":"pith_short_8","alias_value":"YPDCWFK5","created_at":"2026-07-05T01:13:49.462933+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.23489","citing_title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","ref_index":56,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U","json":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U.json","graph_json":"https://pith.science/api/pith-number/YPDCWFK5KQ6LRFUCXY4O7BFD7U/graph.json","events_json":"https://pith.science/api/pith-number/YPDCWFK5KQ6LRFUCXY4O7BFD7U/events.json","paper":"https://pith.science/paper/YPDCWFK5"},"agent_actions":{"view_html":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U","download_json":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U.json","view_paper":"https://pith.science/paper/YPDCWFK5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.15081&json=true","fetch_graph":"https://pith.science/api/pith-number/YPDCWFK5KQ6LRFUCXY4O7BFD7U/graph.json","fetch_events":"https://pith.science/api/pith-number/YPDCWFK5KQ6LRFUCXY4O7BFD7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U/action/storage_attestation","attest_author":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U/action/author_attestation","sign_citation":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U/action/citation_signature","submit_replication":"https://pith.science/pith/YPDCWFK5KQ6LRFUCXY4O7BFD7U/action/replication_record"}},"created_at":"2026-07-05T01:13:49.462933+00:00","updated_at":"2026-07-05T01:13:49.462933+00:00"}