{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LYRIAA3QCFK67ZLW6K3MIPOFJR","short_pith_number":"pith:LYRIAA3Q","schema_version":"1.0","canonical_sha256":"5e228003701155efe576f2b6c43dc54c7057cd1763da1a6cdc8ca59a3314866c","source":{"kind":"arxiv","id":"2108.04620","version":1},"attestation_state":"computed","paper":{"title":"A proof of convergence for the gradient descent optimization method with random initializations in the training of neural networks with ReLU activation for piecewise linear target functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NA","math.NA"],"primary_cat":"math.OC","authors_text":"Adrian Riekert, Arnulf Jentzen","submitted_at":"2021-08-10T12:01:37Z","abstract_excerpt":"Gradient descent (GD) type optimization methods are the standard instrument to train artificial neural networks (ANNs) with rectified linear unit (ReLU) activation. Despite the great success of GD type optimization methods in numerical simulations for the training of ANNs with ReLU activation, it remains - even in the simplest situation of the plain vanilla GD optimization method with random initializations and ANNs with one hidden layer - an open problem to prove (or disprove) the conjecture that the risk of the GD optimization method converges in the training of such ANNs to zero as the widt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.04620","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2021-08-10T12:01:37Z","cross_cats_sorted":["cs.LG","cs.NA","math.NA"],"title_canon_sha256":"14742a7c8251bfb3986cae67cd9eaf289b4b4dd3810a18e0808f2fe11476a867","abstract_canon_sha256":"199c1428ac208284d1b973c99c018cd18e9bdb8eeba7f66389cc1840476770ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:28:42.158376Z","signature_b64":"S70QMvELhZ+NWtgRCcliVM/DjZ1SOfNdKpAPavMV+ZXX4BYqkFLOlOhv59DyxgA3zlhO7nvV1M5W332wT3DmBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e228003701155efe576f2b6c43dc54c7057cd1763da1a6cdc8ca59a3314866c","last_reissued_at":"2026-07-05T05:28:42.157893Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:28:42.157893Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A proof of convergence for the gradient descent optimization method with random initializations in the training of neural networks with ReLU activation for piecewise linear target functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NA","math.NA"],"primary_cat":"math.OC","authors_text":"Adrian Riekert, Arnulf Jentzen","submitted_at":"2021-08-10T12:01:37Z","abstract_excerpt":"Gradient descent (GD) type optimization methods are the standard instrument to train artificial neural networks (ANNs) with rectified linear unit (ReLU) activation. Despite the great success of GD type optimization methods in numerical simulations for the training of ANNs with ReLU activation, it remains - even in the simplest situation of the plain vanilla GD optimization method with random initializations and ANNs with one hidden layer - an open problem to prove (or disprove) the conjecture that the risk of the GD optimization method converges in the training of such ANNs to zero as the widt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.04620","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.04620/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.04620","created_at":"2026-07-05T05:28:42.157953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.04620v1","created_at":"2026-07-05T05:28:42.157953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.04620","created_at":"2026-07-05T05:28:42.157953+00:00"},{"alias_kind":"pith_short_12","alias_value":"LYRIAA3QCFK6","created_at":"2026-07-05T05:28:42.157953+00:00"},{"alias_kind":"pith_short_16","alias_value":"LYRIAA3QCFK67ZLW","created_at":"2026-07-05T05:28:42.157953+00:00"},{"alias_kind":"pith_short_8","alias_value":"LYRIAA3Q","created_at":"2026-07-05T05:28:42.157953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.04607","citing_title":"On MUON optimization: From non-convergence to an error analysis with Polar Express and the Newton-Schulz polynomial from implementations","ref_index":94,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR","json":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR.json","graph_json":"https://pith.science/api/pith-number/LYRIAA3QCFK67ZLW6K3MIPOFJR/graph.json","events_json":"https://pith.science/api/pith-number/LYRIAA3QCFK67ZLW6K3MIPOFJR/events.json","paper":"https://pith.science/paper/LYRIAA3Q"},"agent_actions":{"view_html":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR","download_json":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR.json","view_paper":"https://pith.science/paper/LYRIAA3Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.04620&json=true","fetch_graph":"https://pith.science/api/pith-number/LYRIAA3QCFK67ZLW6K3MIPOFJR/graph.json","fetch_events":"https://pith.science/api/pith-number/LYRIAA3QCFK67ZLW6K3MIPOFJR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR/action/storage_attestation","attest_author":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR/action/author_attestation","sign_citation":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR/action/citation_signature","submit_replication":"https://pith.science/pith/LYRIAA3QCFK67ZLW6K3MIPOFJR/action/replication_record"}},"created_at":"2026-07-05T05:28:42.157953+00:00","updated_at":"2026-07-05T05:28:42.157953+00:00"}