{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U7IQQBAFZIDRO65J7ZY2STAYKE","short_pith_number":"pith:U7IQQBAF","schema_version":"1.0","canonical_sha256":"a7d1080405ca07177ba9fe71a94c185127650d9acfee0e6f888dd7730b7ecc6e","source":{"kind":"arxiv","id":"2402.03982","version":2},"attestation_state":"computed","paper":{"title":"On Convergence of Adam for Stochastic Optimization under Relaxed Assumptions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Junhong Lin, Yusu Hong","submitted_at":"2024-02-06T13:19:26Z","abstract_excerpt":"The Adaptive Momentum Estimation (Adam) algorithm is highly effective in training various deep learning tasks. Despite this, there's limited theoretical understanding for Adam, especially when focusing on its vanilla form in non-convex smooth scenarios with potential unbounded gradients and affine variance noise. In this paper, we study vanilla Adam under these challenging conditions. We introduce a comprehensive noise model which governs affine variance noise, bounded noise and sub-Gaussian noise. We show that Adam can find a stationary point with a $\\mathcal{O}(\\text{poly}(\\log T)/\\sqrt{T})$"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.03982","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-02-06T13:19:26Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"fa08acc58894f463f6c1ecc0caf58a9626c5c11f5e80ab49fd996c47a4a90f14","abstract_canon_sha256":"7b7d9c255f06be701d5eea2bbddcfecaf5c356729884d0db850e94cbd732e9b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:07.685464Z","signature_b64":"ZzYREVh9L4lfnKnKF+wcGb5wa9odK6DP495kqht/573CHhyo13S6Hk+K+RATQfYwYNJPDohqXpp/4zJNpXqsAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7d1080405ca07177ba9fe71a94c185127650d9acfee0e6f888dd7730b7ecc6e","last_reissued_at":"2026-07-05T10:18:07.684997Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:07.684997Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Convergence of Adam for Stochastic Optimization under Relaxed Assumptions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Junhong Lin, Yusu Hong","submitted_at":"2024-02-06T13:19:26Z","abstract_excerpt":"The Adaptive Momentum Estimation (Adam) algorithm is highly effective in training various deep learning tasks. Despite this, there's limited theoretical understanding for Adam, especially when focusing on its vanilla form in non-convex smooth scenarios with potential unbounded gradients and affine variance noise. In this paper, we study vanilla Adam under these challenging conditions. We introduce a comprehensive noise model which governs affine variance noise, bounded noise and sub-Gaussian noise. We show that Adam can find a stationary point with a $\\mathcal{O}(\\text{poly}(\\log T)/\\sqrt{T})$"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.03982","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.03982/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.03982","created_at":"2026-07-05T10:18:07.685065+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.03982v2","created_at":"2026-07-05T10:18:07.685065+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.03982","created_at":"2026-07-05T10:18:07.685065+00:00"},{"alias_kind":"pith_short_12","alias_value":"U7IQQBAFZIDR","created_at":"2026-07-05T10:18:07.685065+00:00"},{"alias_kind":"pith_short_16","alias_value":"U7IQQBAFZIDRO65J","created_at":"2026-07-05T10:18:07.685065+00:00"},{"alias_kind":"pith_short_8","alias_value":"U7IQQBAF","created_at":"2026-07-05T10:18:07.685065+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08783","citing_title":"OptMuon: Closed-Loop Orthogonalized Momentum Methods for Stochastic Optimization with Zero-Noise Optimality","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08783","citing_title":"OptMuon: Closed-Loop Orthogonalized Momentum Methods for Stochastic Optimization with Zero-Noise Optimality","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03099","citing_title":"Why Adam Can Beat SGD: Second-Moment Normalization Yields Sharper Tails","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03099","citing_title":"Why Adam Can Beat SGD: Second-Moment Normalization Yields Sharper Tails","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE","json":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE.json","graph_json":"https://pith.science/api/pith-number/U7IQQBAFZIDRO65J7ZY2STAYKE/graph.json","events_json":"https://pith.science/api/pith-number/U7IQQBAFZIDRO65J7ZY2STAYKE/events.json","paper":"https://pith.science/paper/U7IQQBAF"},"agent_actions":{"view_html":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE","download_json":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE.json","view_paper":"https://pith.science/paper/U7IQQBAF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.03982&json=true","fetch_graph":"https://pith.science/api/pith-number/U7IQQBAFZIDRO65J7ZY2STAYKE/graph.json","fetch_events":"https://pith.science/api/pith-number/U7IQQBAFZIDRO65J7ZY2STAYKE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE/action/storage_attestation","attest_author":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE/action/author_attestation","sign_citation":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE/action/citation_signature","submit_replication":"https://pith.science/pith/U7IQQBAFZIDRO65J7ZY2STAYKE/action/replication_record"}},"created_at":"2026-07-05T10:18:07.685065+00:00","updated_at":"2026-07-05T10:18:07.685065+00:00"}