{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:CDIEAUPYQINYJFZ6ZAWQDJXVPC","short_pith_number":"pith:CDIEAUPY","schema_version":"1.0","canonical_sha256":"10d04051f8821b84973ec82d01a6f5788c8ffd798c068566c64bafa64e262381","source":{"kind":"arxiv","id":"2007.07989","version":2},"attestation_state":"computed","paper":{"title":"An Improved Analysis of Stochastic Gradient Descent with Momentum","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Wotao Yin, Yanli Liu, Yuan Gao","submitted_at":"2020-07-15T20:49:35Z","abstract_excerpt":"SGD with momentum (SGDM) has been widely applied in many machine learning tasks, and it is often applied with dynamic stepsizes and momentum weights tuned in a stagewise manner. Despite of its empirical advantage over SGD, the role of momentum is still unclear in general since previous analyses on SGDM either provide worse convergence bounds than those of SGD, or assume Lipschitz or quadratic objectives, which fail to hold in practice. Furthermore, the role of dynamic parameters has not been addressed. In this work, we show that SGDM converges as fast as SGD for smooth objectives under both st"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.07989","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2020-07-15T20:49:35Z","cross_cats_sorted":[],"title_canon_sha256":"b038f6e1fb0d22e5a9d194e7f157ef41f50bbe6e6eda31a5770a444a133bcd09","abstract_canon_sha256":"9e63ee292ca43432e9ff34f4730aea106b6134bedcddadc3a33f1d9882a09b4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:27:47.885355Z","signature_b64":"6dJ69DULSh7YRbd9fXGN199kOe1Bv5wegW6Caf4bSlR++UCl9rteCzRuev7VXS6TVb9IhZr8NxNXr+4uKovDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"10d04051f8821b84973ec82d01a6f5788c8ffd798c068566c64bafa64e262381","last_reissued_at":"2026-07-05T01:27:47.884925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:27:47.884925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Improved Analysis of Stochastic Gradient Descent with Momentum","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Wotao Yin, Yanli Liu, Yuan Gao","submitted_at":"2020-07-15T20:49:35Z","abstract_excerpt":"SGD with momentum (SGDM) has been widely applied in many machine learning tasks, and it is often applied with dynamic stepsizes and momentum weights tuned in a stagewise manner. Despite of its empirical advantage over SGD, the role of momentum is still unclear in general since previous analyses on SGDM either provide worse convergence bounds than those of SGD, or assume Lipschitz or quadratic objectives, which fail to hold in practice. Furthermore, the role of dynamic parameters has not been addressed. In this work, we show that SGDM converges as fast as SGD for smooth objectives under both st"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.07989","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.07989/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.07989","created_at":"2026-07-05T01:27:47.884981+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.07989v2","created_at":"2026-07-05T01:27:47.884981+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.07989","created_at":"2026-07-05T01:27:47.884981+00:00"},{"alias_kind":"pith_short_12","alias_value":"CDIEAUPYQINY","created_at":"2026-07-05T01:27:47.884981+00:00"},{"alias_kind":"pith_short_16","alias_value":"CDIEAUPYQINYJFZ6","created_at":"2026-07-05T01:27:47.884981+00:00"},{"alias_kind":"pith_short_8","alias_value":"CDIEAUPY","created_at":"2026-07-05T01:27:47.884981+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14345","citing_title":"Convergence of difference inclusions via a diameter criterion","ref_index":177,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC","json":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC.json","graph_json":"https://pith.science/api/pith-number/CDIEAUPYQINYJFZ6ZAWQDJXVPC/graph.json","events_json":"https://pith.science/api/pith-number/CDIEAUPYQINYJFZ6ZAWQDJXVPC/events.json","paper":"https://pith.science/paper/CDIEAUPY"},"agent_actions":{"view_html":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC","download_json":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC.json","view_paper":"https://pith.science/paper/CDIEAUPY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.07989&json=true","fetch_graph":"https://pith.science/api/pith-number/CDIEAUPYQINYJFZ6ZAWQDJXVPC/graph.json","fetch_events":"https://pith.science/api/pith-number/CDIEAUPYQINYJFZ6ZAWQDJXVPC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC/action/storage_attestation","attest_author":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC/action/author_attestation","sign_citation":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC/action/citation_signature","submit_replication":"https://pith.science/pith/CDIEAUPYQINYJFZ6ZAWQDJXVPC/action/replication_record"}},"created_at":"2026-07-05T01:27:47.884981+00:00","updated_at":"2026-07-05T01:27:47.884981+00:00"}