{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:KDL23JR73W5LU5OUYGMGK7EEGJ","short_pith_number":"pith:KDL23JR7","schema_version":"1.0","canonical_sha256":"50d7ada63fddbaba75d4c198657c8432757c35e14c1820b44ca8b61138e52ce8","source":{"kind":"arxiv","id":"1806.01811","version":8},"attestation_state":"computed","paper":{"title":"AdaGrad stepsizes: Sharp convergence over nonconvex landscapes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Leon Bottou, Rachel Ward, Xiaoxia Wu","submitted_at":"2018-06-05T16:59:08Z","abstract_excerpt":"Adaptive gradient methods such as AdaGrad and its variants update the stepsize in stochastic gradient descent on the fly according to the gradients received along the way; such methods have gained widespread use in large-scale optimization for their ability to converge robustly, without the need to fine-tune the stepsize schedule. Yet, the theoretical guarantees to date for AdaGrad are for online and convex optimization. We bridge this gap by providing theoretical guarantees for the convergence of AdaGrad for smooth, nonconvex functions. We show that the norm version of AdaGrad (AdaGrad-Norm) "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1806.01811","kind":"arxiv","version":8},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2018-06-05T16:59:08Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"cfc02d6074a0cc6979e88b5b85c93ab23368d45297b86910ef4ac75a7084f7be","abstract_canon_sha256":"97e19672973f3f97a217c88e2b79af060cdcb731214d0f5cea4b34d5701c88e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:32:43.234267Z","signature_b64":"JcxwH604ZW/FZPkSwDB/J9Ude3cmzuMhZT5RfsqI9U+zoM21CMjle/Mp/XkeRfEK3rtsgITWDzP+crX/BFdlDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50d7ada63fddbaba75d4c198657c8432757c35e14c1820b44ca8b61138e52ce8","last_reissued_at":"2026-07-05T02:32:43.233837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:32:43.233837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaGrad stepsizes: Sharp convergence over nonconvex landscapes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Leon Bottou, Rachel Ward, Xiaoxia Wu","submitted_at":"2018-06-05T16:59:08Z","abstract_excerpt":"Adaptive gradient methods such as AdaGrad and its variants update the stepsize in stochastic gradient descent on the fly according to the gradients received along the way; such methods have gained widespread use in large-scale optimization for their ability to converge robustly, without the need to fine-tune the stepsize schedule. Yet, the theoretical guarantees to date for AdaGrad are for online and convex optimization. We bridge this gap by providing theoretical guarantees for the convergence of AdaGrad for smooth, nonconvex functions. We show that the norm version of AdaGrad (AdaGrad-Norm) "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1806.01811","kind":"arxiv","version":8},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1806.01811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1806.01811","created_at":"2026-07-05T02:32:43.233897+00:00"},{"alias_kind":"arxiv_version","alias_value":"1806.01811v8","created_at":"2026-07-05T02:32:43.233897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1806.01811","created_at":"2026-07-05T02:32:43.233897+00:00"},{"alias_kind":"pith_short_12","alias_value":"KDL23JR73W5L","created_at":"2026-07-05T02:32:43.233897+00:00"},{"alias_kind":"pith_short_16","alias_value":"KDL23JR73W5LU5OU","created_at":"2026-07-05T02:32:43.233897+00:00"},{"alias_kind":"pith_short_8","alias_value":"KDL23JR7","created_at":"2026-07-05T02:32:43.233897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2003.00295","citing_title":"Adaptive Federated Optimization","ref_index":239,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ","json":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ.json","graph_json":"https://pith.science/api/pith-number/KDL23JR73W5LU5OUYGMGK7EEGJ/graph.json","events_json":"https://pith.science/api/pith-number/KDL23JR73W5LU5OUYGMGK7EEGJ/events.json","paper":"https://pith.science/paper/KDL23JR7"},"agent_actions":{"view_html":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ","download_json":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ.json","view_paper":"https://pith.science/paper/KDL23JR7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1806.01811&json=true","fetch_graph":"https://pith.science/api/pith-number/KDL23JR73W5LU5OUYGMGK7EEGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/KDL23JR73W5LU5OUYGMGK7EEGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ/action/storage_attestation","attest_author":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ/action/author_attestation","sign_citation":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ/action/citation_signature","submit_replication":"https://pith.science/pith/KDL23JR73W5LU5OUYGMGK7EEGJ/action/replication_record"}},"created_at":"2026-07-05T02:32:43.233897+00:00","updated_at":"2026-07-05T02:32:43.233897+00:00"}