{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UABD4EBWQN5FWABUGVECCABY2D","short_pith_number":"pith:UABD4EBW","schema_version":"1.0","canonical_sha256":"a0023e1036837a5b00343548210038d0c4aef477e5bf3e0dccac1d4b9bf50181","source":{"kind":"arxiv","id":"2310.17074","version":1},"attestation_state":"computed","paper":{"title":"Benign Oscillation of Stochastic Gradient Descent with Large Learning Rates","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Beining Wu, Difan Zou, Miao Lu, Xiaodong Yang","submitted_at":"2023-10-26T00:35:40Z","abstract_excerpt":"In this work, we theoretically investigate the generalization properties of neural networks (NN) trained by stochastic gradient descent (SGD) algorithm with large learning rates. Under such a training regime, our finding is that, the oscillation of the NN weights caused by the large learning rate SGD training turns out to be beneficial to the generalization of the NN, which potentially improves over the same NN trained by SGD with small learning rates that converges more smoothly. In view of this finding, we call such a phenomenon \"benign oscillation\". Our theory towards demystifying such a ph"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17074","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-26T00:35:40Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"3c4ec1c5e4908a7ce3d7806026bfb39cf97615909b9205baa2394d8c3b931fd4","abstract_canon_sha256":"98fd064c1fcf5bd53dd4674bc706ef738c8303d68eeef8f2e0afe0cdf42a0d4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:19.293345Z","signature_b64":"Fh0LcJvneSgArAiHeORbMF5PopHmF59pHydGQFf8iBjFQo7pWFGIPcBCXMvqgA+2NBbwNLvRzzBwjlIPq5DUDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0023e1036837a5b00343548210038d0c4aef477e5bf3e0dccac1d4b9bf50181","last_reissued_at":"2026-07-05T07:05:19.292940Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:19.292940Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benign Oscillation of Stochastic Gradient Descent with Large Learning Rates","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Beining Wu, Difan Zou, Miao Lu, Xiaodong Yang","submitted_at":"2023-10-26T00:35:40Z","abstract_excerpt":"In this work, we theoretically investigate the generalization properties of neural networks (NN) trained by stochastic gradient descent (SGD) algorithm with large learning rates. Under such a training regime, our finding is that, the oscillation of the NN weights caused by the large learning rate SGD training turns out to be beneficial to the generalization of the NN, which potentially improves over the same NN trained by SGD with small learning rates that converges more smoothly. In view of this finding, we call such a phenomenon \"benign oscillation\". Our theory towards demystifying such a ph"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17074","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17074/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17074","created_at":"2026-07-05T07:05:19.292995+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17074v1","created_at":"2026-07-05T07:05:19.292995+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17074","created_at":"2026-07-05T07:05:19.292995+00:00"},{"alias_kind":"pith_short_12","alias_value":"UABD4EBWQN5F","created_at":"2026-07-05T07:05:19.292995+00:00"},{"alias_kind":"pith_short_16","alias_value":"UABD4EBWQN5FWABU","created_at":"2026-07-05T07:05:19.292995+00:00"},{"alias_kind":"pith_short_8","alias_value":"UABD4EBW","created_at":"2026-07-05T07:05:19.292995+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.08025","citing_title":"Criteria and Bias of Parameterized Linear Regression under Edge of Stability Regime","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D","json":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D.json","graph_json":"https://pith.science/api/pith-number/UABD4EBWQN5FWABUGVECCABY2D/graph.json","events_json":"https://pith.science/api/pith-number/UABD4EBWQN5FWABUGVECCABY2D/events.json","paper":"https://pith.science/paper/UABD4EBW"},"agent_actions":{"view_html":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D","download_json":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D.json","view_paper":"https://pith.science/paper/UABD4EBW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17074&json=true","fetch_graph":"https://pith.science/api/pith-number/UABD4EBWQN5FWABUGVECCABY2D/graph.json","fetch_events":"https://pith.science/api/pith-number/UABD4EBWQN5FWABUGVECCABY2D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D/action/storage_attestation","attest_author":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D/action/author_attestation","sign_citation":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D/action/citation_signature","submit_replication":"https://pith.science/pith/UABD4EBWQN5FWABUGVECCABY2D/action/replication_record"}},"created_at":"2026-07-05T07:05:19.292995+00:00","updated_at":"2026-07-05T07:05:19.292995+00:00"}