{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK","short_pith_number":"pith:3PI3ZB4K","schema_version":"1.0","canonical_sha256":"dbd1bc878af63ec3313ca6a50c27600aa82212eb7045bea0fb720e58e6af06d3","source":{"kind":"arxiv","id":"2112.07369","version":2},"attestation_state":"computed","paper":{"title":"Convergence proof for stochastic gradient descent in the training of deep neural networks with ReLU activation for constant target functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NA","math.NA","math.PR"],"primary_cat":"cs.LG","authors_text":"Adrian Riekert, Arnulf Jentzen, Katharina Pohl, Luca Scarpa, Martin Hutzenthaler","submitted_at":"2021-12-13T11:45:36Z","abstract_excerpt":"In many numerical simulations stochastic gradient descent (SGD) type optimization methods perform very effectively in the training of deep neural networks (DNNs) but till this day it remains an open problem of research to provide a mathematical convergence analysis which rigorously explains the success of SGD type optimization methods in the training of DNNs. In this work we study SGD type optimization methods in the training of fully-connected feedforward DNNs with rectified linear unit (ReLU) activation. We first establish general regularity properties for the risk functions and their genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.07369","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-13T11:45:36Z","cross_cats_sorted":["cs.NA","math.NA","math.PR"],"title_canon_sha256":"7d62b9024fb8c61b502952d95d301aaa1090fdf88060461353923ed749a9c4ab","abstract_canon_sha256":"42b9434fc573a1bcf8314dfc996343d186240a30e586d329d5318767adab728b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:23:53.177743Z","signature_b64":"LF6TuecB5EVrpgeLR4rlb+6iWyrsKol3Gf70VPGlOylIvMVUt/Hbm+JOMsULmrvfOGdzhv4UIhAF9Klx2pFyAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbd1bc878af63ec3313ca6a50c27600aa82212eb7045bea0fb720e58e6af06d3","last_reissued_at":"2026-07-05T06:23:53.177333Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:23:53.177333Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Convergence proof for stochastic gradient descent in the training of deep neural networks with ReLU activation for constant target functions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NA","math.NA","math.PR"],"primary_cat":"cs.LG","authors_text":"Adrian Riekert, Arnulf Jentzen, Katharina Pohl, Luca Scarpa, Martin Hutzenthaler","submitted_at":"2021-12-13T11:45:36Z","abstract_excerpt":"In many numerical simulations stochastic gradient descent (SGD) type optimization methods perform very effectively in the training of deep neural networks (DNNs) but till this day it remains an open problem of research to provide a mathematical convergence analysis which rigorously explains the success of SGD type optimization methods in the training of DNNs. In this work we study SGD type optimization methods in the training of fully-connected feedforward DNNs with rectified linear unit (ReLU) activation. We first establish general regularity properties for the risk functions and their genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.07369","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.07369/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.07369","created_at":"2026-07-05T06:23:53.177392+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.07369v2","created_at":"2026-07-05T06:23:53.177392+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.07369","created_at":"2026-07-05T06:23:53.177392+00:00"},{"alias_kind":"pith_short_12","alias_value":"3PI3ZB4K6Y7M","created_at":"2026-07-05T06:23:53.177392+00:00"},{"alias_kind":"pith_short_16","alias_value":"3PI3ZB4K6Y7MGMJ4","created_at":"2026-07-05T06:23:53.177392+00:00"},{"alias_kind":"pith_short_8","alias_value":"3PI3ZB4K","created_at":"2026-07-05T06:23:53.177392+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.15646","citing_title":"Mathematical analysis of the gradients in deep learning","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK","json":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK.json","graph_json":"https://pith.science/api/pith-number/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/graph.json","events_json":"https://pith.science/api/pith-number/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/events.json","paper":"https://pith.science/paper/3PI3ZB4K"},"agent_actions":{"view_html":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK","download_json":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK.json","view_paper":"https://pith.science/paper/3PI3ZB4K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.07369&json=true","fetch_graph":"https://pith.science/api/pith-number/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/graph.json","fetch_events":"https://pith.science/api/pith-number/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/action/storage_attestation","attest_author":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/action/author_attestation","sign_citation":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/action/citation_signature","submit_replication":"https://pith.science/pith/3PI3ZB4K6Y7MGMJ4U2SQYJ3ABK/action/replication_record"}},"created_at":"2026-07-05T06:23:53.177392+00:00","updated_at":"2026-07-05T06:23:53.177392+00:00"}