{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:YAN6L6S7J3G34GIZMSPFRQRUOJ","short_pith_number":"pith:YAN6L6S7","schema_version":"1.0","canonical_sha256":"c01be5fa5f4ecdbe1919649e58c234727193538bea383f1326e1841db0be7795","source":{"kind":"arxiv","id":"1909.12292","version":4},"attestation_state":"computed","paper":{"title":"Polylogarithmic width suffices for gradient descent to achieve arbitrarily small test error with shallow ReLU networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Matus Telgarsky, Ziwei Ji","submitted_at":"2019-09-26T17:56:28Z","abstract_excerpt":"Recent theoretical work has guaranteed that overparameterized networks trained by gradient descent achieve arbitrarily low training error, and sometimes even low test error. The required width, however, is always polynomial in at least one of the sample size $n$, the (inverse) target error $1/\\epsilon$, and the (inverse) failure probability $1/\\delta$. This work shows that $\\widetilde{\\Theta}(1/\\epsilon)$ iterations of gradient descent with $\\widetilde{\\Omega}(1/\\epsilon^2)$ training examples on two-layer ReLU networks of any width exceeding $\\mathrm{polylog}(n,1/\\epsilon,1/\\delta)$ suffice to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.12292","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-09-26T17:56:28Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"880db42f3546367acb3553da6df67b662300ebe91d8e250f41524d4053c95ddf","abstract_canon_sha256":"ddfa2f07869533a90656aa8656a525815fb85f38cb6680a4d50d4b80d5102533"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:40:58.660405Z","signature_b64":"MXZHiYwNVDj+BLDhso+I5QIv/p+0bXvu5fyxcl6MFnBzHw4pY6MJBSRZqggb0ZaiftFa+33hXwdhW791xSiPBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c01be5fa5f4ecdbe1919649e58c234727193538bea383f1326e1841db0be7795","last_reissued_at":"2026-07-05T00:40:58.659971Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:40:58.659971Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Polylogarithmic width suffices for gradient descent to achieve arbitrarily small test error with shallow ReLU networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Matus Telgarsky, Ziwei Ji","submitted_at":"2019-09-26T17:56:28Z","abstract_excerpt":"Recent theoretical work has guaranteed that overparameterized networks trained by gradient descent achieve arbitrarily low training error, and sometimes even low test error. The required width, however, is always polynomial in at least one of the sample size $n$, the (inverse) target error $1/\\epsilon$, and the (inverse) failure probability $1/\\delta$. This work shows that $\\widetilde{\\Theta}(1/\\epsilon)$ iterations of gradient descent with $\\widetilde{\\Omega}(1/\\epsilon^2)$ training examples on two-layer ReLU networks of any width exceeding $\\mathrm{polylog}(n,1/\\epsilon,1/\\delta)$ suffice to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.12292","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.12292/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.12292","created_at":"2026-07-05T00:40:58.660028+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.12292v4","created_at":"2026-07-05T00:40:58.660028+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.12292","created_at":"2026-07-05T00:40:58.660028+00:00"},{"alias_kind":"pith_short_12","alias_value":"YAN6L6S7J3G3","created_at":"2026-07-05T00:40:58.660028+00:00"},{"alias_kind":"pith_short_16","alias_value":"YAN6L6S7J3G34GIZ","created_at":"2026-07-05T00:40:58.660028+00:00"},{"alias_kind":"pith_short_8","alias_value":"YAN6L6S7","created_at":"2026-07-05T00:40:58.660028+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30831","citing_title":"Geometric Dyson Brownian Motions and the Free Log-Normal Limit for a Non-Square Gaussian Matrix Product","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14345","citing_title":"Convergence of difference inclusions: a diameter criterion and step-size conditions","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ","json":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ.json","graph_json":"https://pith.science/api/pith-number/YAN6L6S7J3G34GIZMSPFRQRUOJ/graph.json","events_json":"https://pith.science/api/pith-number/YAN6L6S7J3G34GIZMSPFRQRUOJ/events.json","paper":"https://pith.science/paper/YAN6L6S7"},"agent_actions":{"view_html":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ","download_json":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ.json","view_paper":"https://pith.science/paper/YAN6L6S7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.12292&json=true","fetch_graph":"https://pith.science/api/pith-number/YAN6L6S7J3G34GIZMSPFRQRUOJ/graph.json","fetch_events":"https://pith.science/api/pith-number/YAN6L6S7J3G34GIZMSPFRQRUOJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ/action/storage_attestation","attest_author":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ/action/author_attestation","sign_citation":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ/action/citation_signature","submit_replication":"https://pith.science/pith/YAN6L6S7J3G34GIZMSPFRQRUOJ/action/replication_record"}},"created_at":"2026-07-05T00:40:58.660028+00:00","updated_at":"2026-07-05T00:40:58.660028+00:00"}