{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DZQUH6C5PWLFGUWWNIBOOL33G7","short_pith_number":"pith:DZQUH6C5","schema_version":"1.0","canonical_sha256":"1e6143f85d7d965352d66a02e72f7b37f66771c8103c64e61acb5b0bd2d36460","source":{"kind":"arxiv","id":"2402.05155","version":1},"attestation_state":"computed","paper":{"title":"Non-convergence to global minimizers for Adam and stochastic gradient descent optimization and constructions of local minimizers in the training of artificial neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Adrian Riekert, Arnulf Jentzen","submitted_at":"2024-02-07T16:14:04Z","abstract_excerpt":"Stochastic gradient descent (SGD) optimization methods such as the plain vanilla SGD method and the popular Adam optimizer are nowadays the method of choice in the training of artificial neural networks (ANNs). Despite the remarkable success of SGD methods in the ANN training in numerical simulations, it remains in essentially all practical relevant scenarios an open problem to rigorously explain why SGD methods seem to succeed to train ANNs. In particular, in most practically relevant supervised learning problems, it seems that SGD methods do with high probability not converge to global minim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05155","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2024-02-07T16:14:04Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e1c2a3913d877849f453c8f61b0c518d9c92eca905dcde76cc5f7308bb5ab2e6","abstract_canon_sha256":"f8509f896df970e5536d0adf02f0a2d975f063b0d1880ce9fea53fdff83425fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:45.497633Z","signature_b64":"69J+bvasheZv3IC6lX7pzpiJzRncux/D7vmWyAYrFBdXnUbqnpqhr3DD8ypq5krH2PmLL1kgpT9ZFMSM3rJtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e6143f85d7d965352d66a02e72f7b37f66771c8103c64e61acb5b0bd2d36460","last_reissued_at":"2026-07-05T07:42:45.497214Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:45.497214Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Non-convergence to global minimizers for Adam and stochastic gradient descent optimization and constructions of local minimizers in the training of artificial neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"math.OC","authors_text":"Adrian Riekert, Arnulf Jentzen","submitted_at":"2024-02-07T16:14:04Z","abstract_excerpt":"Stochastic gradient descent (SGD) optimization methods such as the plain vanilla SGD method and the popular Adam optimizer are nowadays the method of choice in the training of artificial neural networks (ANNs). Despite the remarkable success of SGD methods in the ANN training in numerical simulations, it remains in essentially all practical relevant scenarios an open problem to rigorously explain why SGD methods seem to succeed to train ANNs. In particular, in most practically relevant supervised learning problems, it seems that SGD methods do with high probability not converge to global minim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05155","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05155/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05155","created_at":"2026-07-05T07:42:45.497270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05155v1","created_at":"2026-07-05T07:42:45.497270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05155","created_at":"2026-07-05T07:42:45.497270+00:00"},{"alias_kind":"pith_short_12","alias_value":"DZQUH6C5PWLF","created_at":"2026-07-05T07:42:45.497270+00:00"},{"alias_kind":"pith_short_16","alias_value":"DZQUH6C5PWLFGUWW","created_at":"2026-07-05T07:42:45.497270+00:00"},{"alias_kind":"pith_short_8","alias_value":"DZQUH6C5","created_at":"2026-07-05T07:42:45.497270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04779","citing_title":"Constructive Universal Approximation and Sure Convergence for Multi-Layer Neural Networks","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7","json":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7.json","graph_json":"https://pith.science/api/pith-number/DZQUH6C5PWLFGUWWNIBOOL33G7/graph.json","events_json":"https://pith.science/api/pith-number/DZQUH6C5PWLFGUWWNIBOOL33G7/events.json","paper":"https://pith.science/paper/DZQUH6C5"},"agent_actions":{"view_html":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7","download_json":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7.json","view_paper":"https://pith.science/paper/DZQUH6C5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05155&json=true","fetch_graph":"https://pith.science/api/pith-number/DZQUH6C5PWLFGUWWNIBOOL33G7/graph.json","fetch_events":"https://pith.science/api/pith-number/DZQUH6C5PWLFGUWWNIBOOL33G7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7/action/storage_attestation","attest_author":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7/action/author_attestation","sign_citation":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7/action/citation_signature","submit_replication":"https://pith.science/pith/DZQUH6C5PWLFGUWWNIBOOL33G7/action/replication_record"}},"created_at":"2026-07-05T07:42:45.497270+00:00","updated_at":"2026-07-05T07:42:45.497270+00:00"}