{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CJFVGWKEWC2ORJFDN3XCRSY5TF","short_pith_number":"pith:CJFVGWKE","schema_version":"1.0","canonical_sha256":"124b535944b0b4e8a4a36eee28cb1d995228acdf5151e03959409c3f39013670","source":{"kind":"arxiv","id":"2307.15196","version":2},"attestation_state":"computed","paper":{"title":"The Marginal Value of Momentum for Small Learning Rate SGD","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Kaifeng Lyu, Runzhe Wang, Sadhika Malladi, Tianhao Wang, Zhiyuan Li","submitted_at":"2023-07-27T21:01:26Z","abstract_excerpt":"Momentum is known to accelerate the convergence of gradient descent in strongly convex settings without stochastic gradient noise. In stochastic optimization, such as training neural networks, folklore suggests that momentum may help deep learning optimization by reducing the variance of the stochastic gradient update, but previous theoretical analyses do not find momentum to offer any provable acceleration. Theoretical results in this paper clarify the role of momentum in stochastic settings where the learning rate is small and gradient noise is the dominant source of instability, suggesting "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.15196","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-27T21:01:26Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"6ec90fe8a6cd8d99abe8175523771d0c582b49dc6c26345536bae07282d71097","abstract_canon_sha256":"f55ed6029d27fbbf957ba4e4ca7561d07df3acc4b4ce2ab91c5bbbc34673bf7e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:17.672926Z","signature_b64":"59uuCQ3Oa0K60jLpE5PtgQ5hBjIjZVAMBdTM2FjiW5k2syXiBnUOMIT3wppK09zITct+EMgtMk48LZaCTvuFBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"124b535944b0b4e8a4a36eee28cb1d995228acdf5151e03959409c3f39013670","last_reissued_at":"2026-07-05T08:08:17.672454Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:17.672454Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Marginal Value of Momentum for Small Learning Rate SGD","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Kaifeng Lyu, Runzhe Wang, Sadhika Malladi, Tianhao Wang, Zhiyuan Li","submitted_at":"2023-07-27T21:01:26Z","abstract_excerpt":"Momentum is known to accelerate the convergence of gradient descent in strongly convex settings without stochastic gradient noise. In stochastic optimization, such as training neural networks, folklore suggests that momentum may help deep learning optimization by reducing the variance of the stochastic gradient update, but previous theoretical analyses do not find momentum to offer any provable acceleration. Theoretical results in this paper clarify the role of momentum in stochastic settings where the learning rate is small and gradient noise is the dominant source of instability, suggesting "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.15196","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.15196/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.15196","created_at":"2026-07-05T08:08:17.672510+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.15196v2","created_at":"2026-07-05T08:08:17.672510+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.15196","created_at":"2026-07-05T08:08:17.672510+00:00"},{"alias_kind":"pith_short_12","alias_value":"CJFVGWKEWC2O","created_at":"2026-07-05T08:08:17.672510+00:00"},{"alias_kind":"pith_short_16","alias_value":"CJFVGWKEWC2ORJFD","created_at":"2026-07-05T08:08:17.672510+00:00"},{"alias_kind":"pith_short_8","alias_value":"CJFVGWKE","created_at":"2026-07-05T08:08:17.672510+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19179","citing_title":"Compute Efficiency and Serial Runtime Tradeoffs for Stochastic Momentum Methods","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28961","citing_title":"Dynamics of Stochastic Momentum with Sparse Updates in High Dimensions","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18609","citing_title":"Perfect Parallelization in Mini-Batch SGD with Classical Momentum Acceleration","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14108","citing_title":"Momentum Further Constrains Sharpness at the Edge of Stochastic Stability","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF","json":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF.json","graph_json":"https://pith.science/api/pith-number/CJFVGWKEWC2ORJFDN3XCRSY5TF/graph.json","events_json":"https://pith.science/api/pith-number/CJFVGWKEWC2ORJFDN3XCRSY5TF/events.json","paper":"https://pith.science/paper/CJFVGWKE"},"agent_actions":{"view_html":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF","download_json":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF.json","view_paper":"https://pith.science/paper/CJFVGWKE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.15196&json=true","fetch_graph":"https://pith.science/api/pith-number/CJFVGWKEWC2ORJFDN3XCRSY5TF/graph.json","fetch_events":"https://pith.science/api/pith-number/CJFVGWKEWC2ORJFDN3XCRSY5TF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF/action/storage_attestation","attest_author":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF/action/author_attestation","sign_citation":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF/action/citation_signature","submit_replication":"https://pith.science/pith/CJFVGWKEWC2ORJFDN3XCRSY5TF/action/replication_record"}},"created_at":"2026-07-05T08:08:17.672510+00:00","updated_at":"2026-07-05T08:08:17.672510+00:00"}