{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:X4SJ44TPMPPRVGU7H4DADKTG6I","short_pith_number":"pith:X4SJ44TP","schema_version":"1.0","canonical_sha256":"bf249e726f63df1a9a9f3f0601aa66f21e9b0b0fe96fc7ff838e4d23af5bb64f","source":{"kind":"arxiv","id":"2007.14294","version":1},"attestation_state":"computed","paper":{"title":"A High Probability Analysis of Adaptive SGD with Momentum","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Francesco Orabona, Xiaoyu Li","submitted_at":"2020-07-28T15:06:22Z","abstract_excerpt":"Stochastic Gradient Descent (SGD) and its variants are the most used algorithms in machine learning applications. In particular, SGD with adaptive learning rates and momentum is the industry standard to train deep networks. Despite the enormous success of these methods, our theoretical understanding of these variants in the nonconvex setting is not complete, with most of the results only proving convergence in expectation and with strong assumptions on the stochastic gradients. In this paper, we present a high probability analysis for adaptive and momentum algorithms, under weak assumptions on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.14294","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2020-07-28T15:06:22Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"542bc5700b4a48920fac21f362de067b304e2cadea39396d6fafe4e9aa51860d","abstract_canon_sha256":"99fb11557b5e62b423496f00450ebcb022315ddcf3ae6201c5d04a28b80a66ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:22:57.407058Z","signature_b64":"PTOWnrxuUXIuIxonO/DIa+BDhCDwhbTX05PJTmh8kPbKiXKQUfBnv23CZeNOVWvZbC8c1tV6yLUoSDm7EYfgCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf249e726f63df1a9a9f3f0601aa66f21e9b0b0fe96fc7ff838e4d23af5bb64f","last_reissued_at":"2026-07-05T01:22:57.406607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:22:57.406607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A High Probability Analysis of Adaptive SGD with Momentum","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Francesco Orabona, Xiaoyu Li","submitted_at":"2020-07-28T15:06:22Z","abstract_excerpt":"Stochastic Gradient Descent (SGD) and its variants are the most used algorithms in machine learning applications. In particular, SGD with adaptive learning rates and momentum is the industry standard to train deep networks. Despite the enormous success of these methods, our theoretical understanding of these variants in the nonconvex setting is not complete, with most of the results only proving convergence in expectation and with strong assumptions on the stochastic gradients. In this paper, we present a high probability analysis for adaptive and momentum algorithms, under weak assumptions on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.14294","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.14294/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.14294","created_at":"2026-07-05T01:22:57.406662+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.14294v1","created_at":"2026-07-05T01:22:57.406662+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.14294","created_at":"2026-07-05T01:22:57.406662+00:00"},{"alias_kind":"pith_short_12","alias_value":"X4SJ44TPMPPR","created_at":"2026-07-05T01:22:57.406662+00:00"},{"alias_kind":"pith_short_16","alias_value":"X4SJ44TPMPPRVGU7","created_at":"2026-07-05T01:22:57.406662+00:00"},{"alias_kind":"pith_short_8","alias_value":"X4SJ44TP","created_at":"2026-07-05T01:22:57.406662+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17526","citing_title":"MGUP: A Momentum-Gradient Alignment Update Policy for Stochastic Optimization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08783","citing_title":"OptMuon: Closed-Loop Orthogonalized Momentum Methods for Stochastic Optimization with Zero-Noise Optimality","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02701","citing_title":"Robust and Fast Training via Per-Sample Clipping","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08783","citing_title":"OptMuon: Closed-Loop Orthogonalized Momentum Methods for Stochastic Optimization with Zero-Noise Optimality","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I","json":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I.json","graph_json":"https://pith.science/api/pith-number/X4SJ44TPMPPRVGU7H4DADKTG6I/graph.json","events_json":"https://pith.science/api/pith-number/X4SJ44TPMPPRVGU7H4DADKTG6I/events.json","paper":"https://pith.science/paper/X4SJ44TP"},"agent_actions":{"view_html":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I","download_json":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I.json","view_paper":"https://pith.science/paper/X4SJ44TP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.14294&json=true","fetch_graph":"https://pith.science/api/pith-number/X4SJ44TPMPPRVGU7H4DADKTG6I/graph.json","fetch_events":"https://pith.science/api/pith-number/X4SJ44TPMPPRVGU7H4DADKTG6I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I/action/storage_attestation","attest_author":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I/action/author_attestation","sign_citation":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I/action/citation_signature","submit_replication":"https://pith.science/pith/X4SJ44TPMPPRVGU7H4DADKTG6I/action/replication_record"}},"created_at":"2026-07-05T01:22:57.406662+00:00","updated_at":"2026-07-05T01:22:57.406662+00:00"}