{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YBUXY3XUCR5QQLRO6FSHF2CUB7","short_pith_number":"pith:YBUXY3XU","schema_version":"1.0","canonical_sha256":"c0697c6ef4147b082e2ef16472e8540ff9032aa58c8b9ec20e4aa7e17f928fd5","source":{"kind":"arxiv","id":"2406.04592","version":3},"attestation_state":"computed","paper":{"title":"Provable Complexity Improvement of AdaGrad over SGD: Upper and Lower Bounds in Stochastic Non-Convex Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Aryan Mokhtari, Devyani Maladkar, Ruichen Jiang","submitted_at":"2024-06-07T02:55:57Z","abstract_excerpt":"Adaptive gradient methods, such as AdaGrad, are among the most successful optimization algorithms for neural network training. While these methods are known to achieve better dimensional dependence than stochastic gradient descent (SGD) for stochastic convex optimization under favorable geometry, the theoretical justification for their success in stochastic non-convex optimization remains elusive. In fact, under standard assumptions of Lipschitz gradients and bounded noise variance, it is known that SGD is worst-case optimal in terms of finding a near-stationary point with respect to the $l_2$"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04592","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-06-07T02:55:57Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"a3e4dc679f11092440789ce090d029aff24fa83caf2dc172b73450a5c3cff012","abstract_canon_sha256":"18c5c3778c35339e30f710b189501fa6f02cae6cf4a25898c5e30f343ef20fce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:49.395437Z","signature_b64":"2jp0Da30pmgYlnuCTdBexwvRKU48PYMiqxQNmDFCRSqFGMEPFTjnjXXVY3W2hfRkDFbR0+ooi0nGIMYQ7wWuAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0697c6ef4147b082e2ef16472e8540ff9032aa58c8b9ec20e4aa7e17f928fd5","last_reissued_at":"2026-07-05T11:16:49.394832Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:49.394832Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Provable Complexity Improvement of AdaGrad over SGD: Upper and Lower Bounds in Stochastic Non-Convex Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"math.OC","authors_text":"Aryan Mokhtari, Devyani Maladkar, Ruichen Jiang","submitted_at":"2024-06-07T02:55:57Z","abstract_excerpt":"Adaptive gradient methods, such as AdaGrad, are among the most successful optimization algorithms for neural network training. While these methods are known to achieve better dimensional dependence than stochastic gradient descent (SGD) for stochastic convex optimization under favorable geometry, the theoretical justification for their success in stochastic non-convex optimization remains elusive. In fact, under standard assumptions of Lipschitz gradients and bounded noise variance, it is known that SGD is worst-case optimal in terms of finding a near-stationary point with respect to the $l_2$"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04592","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04592","created_at":"2026-07-05T11:16:49.394895+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04592v3","created_at":"2026-07-05T11:16:49.394895+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04592","created_at":"2026-07-05T11:16:49.394895+00:00"},{"alias_kind":"pith_short_12","alias_value":"YBUXY3XUCR5Q","created_at":"2026-07-05T11:16:49.394895+00:00"},{"alias_kind":"pith_short_16","alias_value":"YBUXY3XUCR5QQLRO","created_at":"2026-07-05T11:16:49.394895+00:00"},{"alias_kind":"pith_short_8","alias_value":"YBUXY3XU","created_at":"2026-07-05T11:16:49.394895+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02505","citing_title":"Optimal Projection-Free Adaptive SGD for Matrix Optimization","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06654","citing_title":"Optimizer-Model Consistency: Full Finetuning with the Same Optimizer as Pretraining Forgets Less","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17423","citing_title":"A unified convergence theory for adaptive first-order methods in the nonconvex case, including AdaNorm, full and diagonal AdaGrad, Shampoo and Muo","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7","json":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7.json","graph_json":"https://pith.science/api/pith-number/YBUXY3XUCR5QQLRO6FSHF2CUB7/graph.json","events_json":"https://pith.science/api/pith-number/YBUXY3XUCR5QQLRO6FSHF2CUB7/events.json","paper":"https://pith.science/paper/YBUXY3XU"},"agent_actions":{"view_html":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7","download_json":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7.json","view_paper":"https://pith.science/paper/YBUXY3XU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04592&json=true","fetch_graph":"https://pith.science/api/pith-number/YBUXY3XUCR5QQLRO6FSHF2CUB7/graph.json","fetch_events":"https://pith.science/api/pith-number/YBUXY3XUCR5QQLRO6FSHF2CUB7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7/action/storage_attestation","attest_author":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7/action/author_attestation","sign_citation":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7/action/citation_signature","submit_replication":"https://pith.science/pith/YBUXY3XUCR5QQLRO6FSHF2CUB7/action/replication_record"}},"created_at":"2026-07-05T11:16:49.394895+00:00","updated_at":"2026-07-05T11:16:49.394895+00:00"}