{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:5IZ5ASO45UV4DXSX3I6MQPEV3M","short_pith_number":"pith:5IZ5ASO4","schema_version":"1.0","canonical_sha256":"ea33d049dced2bc1de57da3cc83c95db180c43ed3a426b3679793c4fc410aee4","source":{"kind":"arxiv","id":"1802.06175","version":2},"attestation_state":"computed","paper":{"title":"An Alternative View: When Does SGD Escape Local Minima?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Robert Kleinberg, Yang Yuan, Yuanzhi Li","submitted_at":"2018-02-17T02:44:16Z","abstract_excerpt":"Stochastic gradient descent (SGD) is widely used in machine learning. Although being commonly viewed as a fast but not accurate version of gradient descent (GD), it always finds better solutions than GD for modern neural networks.\n  In order to understand this phenomenon, we take an alternative view that SGD is working on the convolved (thus smoothed) version of the loss function. We show that, even if the function $f$ has many bad local minima or saddle points, as long as for every point $x$, the weighted average of the gradients of its neighborhoods is one point convex with respect to the de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1802.06175","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-02-17T02:44:16Z","cross_cats_sorted":[],"title_canon_sha256":"3a7822bb0e2b03468c1edde49ac113fa59c6e818485a94cfaba522288c107925","abstract_canon_sha256":"4c9920587ff4b5e8eb2688d26fb0a496140acb93e807e803850b4f777f314abd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:08:00.566311Z","signature_b64":"5+rs+0IFBDsiZ9z+jeHditgusJQFlu6oP9/DS38tulUudQfEYYhBXjpDh+q51w6fM5fpHqtuOPzhfF2YH/F5BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea33d049dced2bc1de57da3cc83c95db180c43ed3a426b3679793c4fc410aee4","last_reissued_at":"2026-05-18T00:08:00.565779Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:08:00.565779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Alternative View: When Does SGD Escape Local Minima?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Robert Kleinberg, Yang Yuan, Yuanzhi Li","submitted_at":"2018-02-17T02:44:16Z","abstract_excerpt":"Stochastic gradient descent (SGD) is widely used in machine learning. Although being commonly viewed as a fast but not accurate version of gradient descent (GD), it always finds better solutions than GD for modern neural networks.\n  In order to understand this phenomenon, we take an alternative view that SGD is working on the convolved (thus smoothed) version of the loss function. We show that, even if the function $f$ has many bad local minima or saddle points, as long as for every point $x$, the weighted average of the gradients of its neighborhoods is one point convex with respect to the de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1802.06175","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1802.06175","created_at":"2026-05-18T00:08:00.565864+00:00"},{"alias_kind":"arxiv_version","alias_value":"1802.06175v2","created_at":"2026-05-18T00:08:00.565864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1802.06175","created_at":"2026-05-18T00:08:00.565864+00:00"},{"alias_kind":"pith_short_12","alias_value":"5IZ5ASO45UV4","created_at":"2026-05-18T12:32:08.215937+00:00"},{"alias_kind":"pith_short_16","alias_value":"5IZ5ASO45UV4DXSX","created_at":"2026-05-18T12:32:08.215937+00:00"},{"alias_kind":"pith_short_8","alias_value":"5IZ5ASO4","created_at":"2026-05-18T12:32:08.215937+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"1908.10920","citing_title":"Deep Learning Theory Review: An Optimal Control and Dynamical Systems Perspective","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M","json":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M.json","graph_json":"https://pith.science/api/pith-number/5IZ5ASO45UV4DXSX3I6MQPEV3M/graph.json","events_json":"https://pith.science/api/pith-number/5IZ5ASO45UV4DXSX3I6MQPEV3M/events.json","paper":"https://pith.science/paper/5IZ5ASO4"},"agent_actions":{"view_html":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M","download_json":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M.json","view_paper":"https://pith.science/paper/5IZ5ASO4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1802.06175&json=true","fetch_graph":"https://pith.science/api/pith-number/5IZ5ASO45UV4DXSX3I6MQPEV3M/graph.json","fetch_events":"https://pith.science/api/pith-number/5IZ5ASO45UV4DXSX3I6MQPEV3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M/action/storage_attestation","attest_author":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M/action/author_attestation","sign_citation":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M/action/citation_signature","submit_replication":"https://pith.science/pith/5IZ5ASO45UV4DXSX3I6MQPEV3M/action/replication_record"}},"created_at":"2026-05-18T00:08:00.565864+00:00","updated_at":"2026-05-18T00:08:00.565864+00:00"}