{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:DFJTJNOSIEZPT23DVN7IBDJ6HE","short_pith_number":"pith:DFJTJNOS","schema_version":"1.0","canonical_sha256":"195334b5d24132f9eb63ab7e808d3e3918f6a9c263f73e280946d8d08ddafb1a","source":{"kind":"arxiv","id":"2004.13146","version":1},"attestation_state":"computed","paper":{"title":"The Impact of the Mini-batch Size on the Variance of Gradients in Stochastic Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Diego Klabjan, Xin Qian","submitted_at":"2020-04-27T20:06:11Z","abstract_excerpt":"The mini-batch stochastic gradient descent (SGD) algorithm is widely used in training machine learning models, in particular deep learning models. We study SGD dynamics under linear regression and two-layer linear networks, with an easy extension to deeper linear networks, by focusing on the variance of the gradients, which is the first study of this nature. In the linear regression case, we show that in each iteration the norm of the gradient is a decreasing function of the mini-batch size $b$ and thus the variance of the stochastic gradient estimator is a decreasing function of $b$. For deep"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.13146","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2020-04-27T20:06:11Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ff5d4d38ac8a6dd9f92daecfffbae20acaf6b15f052a59affa20d677ac7b3882","abstract_canon_sha256":"922d8268ee6a974e34452c139f767dd847957fa997d09ff741f31a04008a63e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:58:55.901803Z","signature_b64":"5P6/NEKEZ2nGfF95ks9ouI3sFoKW03LrP+Z0RqlTvEFB0vZoQUe4mVSIU8hCcD3z0FNPg51oRVlUDjG5G1c9CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"195334b5d24132f9eb63ab7e808d3e3918f6a9c263f73e280946d8d08ddafb1a","last_reissued_at":"2026-07-05T00:58:55.901369Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:58:55.901369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Impact of the Mini-batch Size on the Variance of Gradients in Stochastic Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Diego Klabjan, Xin Qian","submitted_at":"2020-04-27T20:06:11Z","abstract_excerpt":"The mini-batch stochastic gradient descent (SGD) algorithm is widely used in training machine learning models, in particular deep learning models. We study SGD dynamics under linear regression and two-layer linear networks, with an easy extension to deeper linear networks, by focusing on the variance of the gradients, which is the first study of this nature. In the linear regression case, we show that in each iteration the norm of the gradient is a decreasing function of the mini-batch size $b$ and thus the variance of the stochastic gradient estimator is a decreasing function of $b$. For deep"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.13146","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.13146/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.13146","created_at":"2026-07-05T00:58:55.901426+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.13146v1","created_at":"2026-07-05T00:58:55.901426+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.13146","created_at":"2026-07-05T00:58:55.901426+00:00"},{"alias_kind":"pith_short_12","alias_value":"DFJTJNOSIEZP","created_at":"2026-07-05T00:58:55.901426+00:00"},{"alias_kind":"pith_short_16","alias_value":"DFJTJNOSIEZPT23D","created_at":"2026-07-05T00:58:55.901426+00:00"},{"alias_kind":"pith_short_8","alias_value":"DFJTJNOS","created_at":"2026-07-05T00:58:55.901426+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17340","citing_title":"Olivia: Harmonizing Time Series Foundation Models with Power Spectral Density","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06350","citing_title":"Convergence of Riemannian Stochastic Gradient Descents: Varying Batch Sizes And Nonstandard Batch Forming","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE","json":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE.json","graph_json":"https://pith.science/api/pith-number/DFJTJNOSIEZPT23DVN7IBDJ6HE/graph.json","events_json":"https://pith.science/api/pith-number/DFJTJNOSIEZPT23DVN7IBDJ6HE/events.json","paper":"https://pith.science/paper/DFJTJNOS"},"agent_actions":{"view_html":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE","download_json":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE.json","view_paper":"https://pith.science/paper/DFJTJNOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.13146&json=true","fetch_graph":"https://pith.science/api/pith-number/DFJTJNOSIEZPT23DVN7IBDJ6HE/graph.json","fetch_events":"https://pith.science/api/pith-number/DFJTJNOSIEZPT23DVN7IBDJ6HE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE/action/storage_attestation","attest_author":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE/action/author_attestation","sign_citation":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE/action/citation_signature","submit_replication":"https://pith.science/pith/DFJTJNOSIEZPT23DVN7IBDJ6HE/action/replication_record"}},"created_at":"2026-07-05T00:58:55.901426+00:00","updated_at":"2026-07-05T00:58:55.901426+00:00"}