{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:BEFXZ2WZRGUEQWHH5DIULSFNAJ","short_pith_number":"pith:BEFXZ2WZ","schema_version":"1.0","canonical_sha256":"090b7cead989a84858e7e8d145c8ad025796660aacc2a926911e33e42924823c","source":{"kind":"arxiv","id":"1902.07111","version":2},"attestation_state":"computed","paper":{"title":"Global Convergence of Adaptive Gradient Methods for An Over-parameterized Neural Network","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Rachel Ward, Simon S. Du, Xiaoxia Wu","submitted_at":"2019-02-19T16:08:55Z","abstract_excerpt":"Adaptive gradient methods like AdaGrad are widely used in optimizing neural networks. Yet, existing convergence guarantees for adaptive gradient methods require either convexity or smoothness, and, in the smooth setting, only guarantee convergence to a stationary point. We propose an adaptive gradient method and show that for two-layer over-parameterized neural networks -- if the width is sufficiently large (polynomially) -- then the proposed method converges \\emph{to the global minimum} in polynomial time, and convergence is robust, \\emph{ without the need to fine-tune hyper-parameters such a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1902.07111","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-02-19T16:08:55Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"3e2f7b0c6c7b3b4afa7a88f2e009daf1a46aca2588d6560fa124f62abeba2b82","abstract_canon_sha256":"b3dc6363e3bd3a7eeaddc3c2a8b4f3de5d300f095e8482c587630d042c223b8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:13:13.536538Z","signature_b64":"aXd4qnWipvA01KyL37PnOFRqe+Q3E31kmgIwU6IPTbhCe8zmTm8xlSwa9x+ebAKnaFLyVMHhCSOdJRvlj0P4CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"090b7cead989a84858e7e8d145c8ad025796660aacc2a926911e33e42924823c","last_reissued_at":"2026-07-05T00:13:13.536007Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:13:13.536007Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Global Convergence of Adaptive Gradient Methods for An Over-parameterized Neural Network","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Rachel Ward, Simon S. Du, Xiaoxia Wu","submitted_at":"2019-02-19T16:08:55Z","abstract_excerpt":"Adaptive gradient methods like AdaGrad are widely used in optimizing neural networks. Yet, existing convergence guarantees for adaptive gradient methods require either convexity or smoothness, and, in the smooth setting, only guarantee convergence to a stationary point. We propose an adaptive gradient method and show that for two-layer over-parameterized neural networks -- if the width is sufficiently large (polynomially) -- then the proposed method converges \\emph{to the global minimum} in polynomial time, and convergence is robust, \\emph{ without the need to fine-tune hyper-parameters such a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1902.07111","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1902.07111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1902.07111","created_at":"2026-07-05T00:13:13.536080+00:00"},{"alias_kind":"arxiv_version","alias_value":"1902.07111v2","created_at":"2026-07-05T00:13:13.536080+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1902.07111","created_at":"2026-07-05T00:13:13.536080+00:00"},{"alias_kind":"pith_short_12","alias_value":"BEFXZ2WZRGUE","created_at":"2026-07-05T00:13:13.536080+00:00"},{"alias_kind":"pith_short_16","alias_value":"BEFXZ2WZRGUEQWHH","created_at":"2026-07-05T00:13:13.536080+00:00"},{"alias_kind":"pith_short_8","alias_value":"BEFXZ2WZ","created_at":"2026-07-05T00:13:13.536080+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27142","citing_title":"Estimation of High Dimensional Bounded Discrete Graphical Models via Regularized Generalized Score Matching","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10089","citing_title":"A Theory on Flow Matching with Neural Networks","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2003.00295","citing_title":"Adaptive Federated Optimization","ref_index":240,"is_internal_anchor":false},{"citing_arxiv_id":"2401.01335","citing_title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ","json":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ.json","graph_json":"https://pith.science/api/pith-number/BEFXZ2WZRGUEQWHH5DIULSFNAJ/graph.json","events_json":"https://pith.science/api/pith-number/BEFXZ2WZRGUEQWHH5DIULSFNAJ/events.json","paper":"https://pith.science/paper/BEFXZ2WZ"},"agent_actions":{"view_html":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ","download_json":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ.json","view_paper":"https://pith.science/paper/BEFXZ2WZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1902.07111&json=true","fetch_graph":"https://pith.science/api/pith-number/BEFXZ2WZRGUEQWHH5DIULSFNAJ/graph.json","fetch_events":"https://pith.science/api/pith-number/BEFXZ2WZRGUEQWHH5DIULSFNAJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ/action/storage_attestation","attest_author":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ/action/author_attestation","sign_citation":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ/action/citation_signature","submit_replication":"https://pith.science/pith/BEFXZ2WZRGUEQWHH5DIULSFNAJ/action/replication_record"}},"created_at":"2026-07-05T00:13:13.536080+00:00","updated_at":"2026-07-05T00:13:13.536080+00:00"}