{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:YJGNZ3J7BMVXOXPME246CS265J","short_pith_number":"pith:YJGNZ3J7","schema_version":"1.0","canonical_sha256":"c24cdced3f0b2b775dec26b9e14b5eea66eee3c4e150feed5793de96461f8479","source":{"kind":"arxiv","id":"1902.04811","version":2},"attestation_state":"computed","paper":{"title":"On Nonconvex Optimization for Machine Learning: Gradients, Stochasticity, and Saddle Points","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Michael I. Jordan, Praneeth Netrapalli, Rong Ge, Sham M. Kakade","submitted_at":"2019-02-13T09:44:02Z","abstract_excerpt":"Gradient descent (GD) and stochastic gradient descent (SGD) are the workhorses of large-scale machine learning. While classical theory focused on analyzing the performance of these methods in convex optimization problems, the most notable successes in machine learning have involved nonconvex optimization, and a gap has arisen between theory and practice. Indeed, traditional analyses of GD and SGD show that both algorithms converge to stationary points efficiently. But these analyses do not take into account the possibility of converging to saddle points. More recent theory has shown that GD an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1902.04811","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-02-13T09:44:02Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"982a5c982193b84f10afe9d2b9c3578aa829c39e28753d167db422f394554008","abstract_canon_sha256":"a23b5866a71a9ec3f309af673a964c2bdc3c4afeb68a2b44b897d0a508887ee7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:12.595115Z","signature_b64":"uJ2mJO9Rgo3sQ3BVt2/9ddA3vqWo/YbepEOSmWjYdCL4+KkmKoohboBhS497lbXptxCkYxVD+fpw4iecRyA8Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c24cdced3f0b2b775dec26b9e14b5eea66eee3c4e150feed5793de96461f8479","last_reissued_at":"2026-07-05T00:02:12.594629Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:12.594629Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Nonconvex Optimization for Machine Learning: Gradients, Stochasticity, and Saddle Points","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Michael I. Jordan, Praneeth Netrapalli, Rong Ge, Sham M. Kakade","submitted_at":"2019-02-13T09:44:02Z","abstract_excerpt":"Gradient descent (GD) and stochastic gradient descent (SGD) are the workhorses of large-scale machine learning. While classical theory focused on analyzing the performance of these methods in convex optimization problems, the most notable successes in machine learning have involved nonconvex optimization, and a gap has arisen between theory and practice. Indeed, traditional analyses of GD and SGD show that both algorithms converge to stationary points efficiently. But these analyses do not take into account the possibility of converging to saddle points. More recent theory has shown that GD an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1902.04811","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1902.04811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1902.04811","created_at":"2026-07-05T00:02:12.594695+00:00"},{"alias_kind":"arxiv_version","alias_value":"1902.04811v2","created_at":"2026-07-05T00:02:12.594695+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1902.04811","created_at":"2026-07-05T00:02:12.594695+00:00"},{"alias_kind":"pith_short_12","alias_value":"YJGNZ3J7BMVX","created_at":"2026-07-05T00:02:12.594695+00:00"},{"alias_kind":"pith_short_16","alias_value":"YJGNZ3J7BMVXOXPM","created_at":"2026-07-05T00:02:12.594695+00:00"},{"alias_kind":"pith_short_8","alias_value":"YJGNZ3J7","created_at":"2026-07-05T00:02:12.594695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05438","citing_title":"Sharp First-Order Lower Bounds for Higher-Order Smooth Nonconvex Optimization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"1907.01848","citing_title":"Distributed Learning in Non-Convex Environments -- Part I: Agreement at a Linear Rate","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"1907.01849","citing_title":"Distributed Learning in Non-Convex Environments -- Part II: Polynomial Escape from Saddle-Points","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14345","citing_title":"Convergence of difference inclusions via a diameter criterion","ref_index":291,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J","json":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J.json","graph_json":"https://pith.science/api/pith-number/YJGNZ3J7BMVXOXPME246CS265J/graph.json","events_json":"https://pith.science/api/pith-number/YJGNZ3J7BMVXOXPME246CS265J/events.json","paper":"https://pith.science/paper/YJGNZ3J7"},"agent_actions":{"view_html":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J","download_json":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J.json","view_paper":"https://pith.science/paper/YJGNZ3J7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1902.04811&json=true","fetch_graph":"https://pith.science/api/pith-number/YJGNZ3J7BMVXOXPME246CS265J/graph.json","fetch_events":"https://pith.science/api/pith-number/YJGNZ3J7BMVXOXPME246CS265J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J/action/storage_attestation","attest_author":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J/action/author_attestation","sign_citation":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J/action/citation_signature","submit_replication":"https://pith.science/pith/YJGNZ3J7BMVXOXPME246CS265J/action/replication_record"}},"created_at":"2026-07-05T00:02:12.594695+00:00","updated_at":"2026-07-05T00:02:12.594695+00:00"}