{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XENUDN3EDV2TG7OGNQCOHVVCGA","short_pith_number":"pith:XENUDN3E","schema_version":"1.0","canonical_sha256":"b91b41b7641d75337dc66c04e3d6a23035e3bf5bfede5317cf3aebbc37988881","source":{"kind":"arxiv","id":"2112.09684","version":2},"attestation_state":"computed","paper":{"title":"On the existence of global minima and convergence analyses for gradient descent methods in the training of deep neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NA","math.NA","math.ST","stat.TH"],"primary_cat":"math.OC","authors_text":"Adrian Riekert, Arnulf Jentzen","submitted_at":"2021-12-17T18:55:40Z","abstract_excerpt":"In this article we study fully-connected feedforward deep ReLU ANNs with an arbitrarily large number of hidden layers and we prove convergence of the risk of the GD optimization method with random initializations in the training of such ANNs under the assumption that the unnormalized probability density function of the probability distribution of the input data of the considered supervised learning problem is piecewise polynomial, under the assumption that the target function (describing the relationship between input data and the output data) is piecewise polynomial, and under the assumption "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.09684","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2021-12-17T18:55:40Z","cross_cats_sorted":["cs.LG","cs.NA","math.NA","math.ST","stat.TH"],"title_canon_sha256":"2603087c5ac9a7900cb7b1ffc9925a9adf545336d54a9bf35f969d25becaf409","abstract_canon_sha256":"d48f60e8aaca009464e7541df651ae6fb542781c20e1500cda2ce9924203067b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:40:00.889348Z","signature_b64":"t6WWhVpCOYC+Ihx7Rh4m/VBtB2rbCQPTN9825+sdNRr5DYwQb69gIygr4hW0RQDCEKf6UBm2cP60L2ep78fyBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b91b41b7641d75337dc66c04e3d6a23035e3bf5bfede5317cf3aebbc37988881","last_reissued_at":"2026-07-05T04:40:00.888897Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:40:00.888897Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the existence of global minima and convergence analyses for gradient descent methods in the training of deep neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NA","math.NA","math.ST","stat.TH"],"primary_cat":"math.OC","authors_text":"Adrian Riekert, Arnulf Jentzen","submitted_at":"2021-12-17T18:55:40Z","abstract_excerpt":"In this article we study fully-connected feedforward deep ReLU ANNs with an arbitrarily large number of hidden layers and we prove convergence of the risk of the GD optimization method with random initializations in the training of such ANNs under the assumption that the unnormalized probability density function of the probability distribution of the input data of the considered supervised learning problem is piecewise polynomial, under the assumption that the target function (describing the relationship between input data and the output data) is piecewise polynomial, and under the assumption "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.09684","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.09684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.09684","created_at":"2026-07-05T04:40:00.888944+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.09684v2","created_at":"2026-07-05T04:40:00.888944+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.09684","created_at":"2026-07-05T04:40:00.888944+00:00"},{"alias_kind":"pith_short_12","alias_value":"XENUDN3EDV2T","created_at":"2026-07-05T04:40:00.888944+00:00"},{"alias_kind":"pith_short_16","alias_value":"XENUDN3EDV2TG7OG","created_at":"2026-07-05T04:40:00.888944+00:00"},{"alias_kind":"pith_short_8","alias_value":"XENUDN3E","created_at":"2026-07-05T04:40:00.888944+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.15646","citing_title":"Mathematical analysis of the gradients in deep learning","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA","json":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA.json","graph_json":"https://pith.science/api/pith-number/XENUDN3EDV2TG7OGNQCOHVVCGA/graph.json","events_json":"https://pith.science/api/pith-number/XENUDN3EDV2TG7OGNQCOHVVCGA/events.json","paper":"https://pith.science/paper/XENUDN3E"},"agent_actions":{"view_html":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA","download_json":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA.json","view_paper":"https://pith.science/paper/XENUDN3E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.09684&json=true","fetch_graph":"https://pith.science/api/pith-number/XENUDN3EDV2TG7OGNQCOHVVCGA/graph.json","fetch_events":"https://pith.science/api/pith-number/XENUDN3EDV2TG7OGNQCOHVVCGA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA/action/storage_attestation","attest_author":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA/action/author_attestation","sign_citation":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA/action/citation_signature","submit_replication":"https://pith.science/pith/XENUDN3EDV2TG7OGNQCOHVVCGA/action/replication_record"}},"created_at":"2026-07-05T04:40:00.888944+00:00","updated_at":"2026-07-05T04:40:00.888944+00:00"}