{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:KKAQV2RE3L6WGALAEY3DQ2F4RU","short_pith_number":"pith:KKAQV2RE","schema_version":"1.0","canonical_sha256":"52810aea24dafd63016026363868bc8d3fcde039759dffbb568f9a56a35cbdeb","source":{"kind":"arxiv","id":"1911.01413","version":3},"attestation_state":"computed","paper":{"title":"Sub-Optimal Local Minima Exist for Neural Networks with Almost All Non-Linear Activations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Dawei Li, Ruoyu Sun, Tian Ding","submitted_at":"2019-11-04T18:56:58Z","abstract_excerpt":"Does over-parameterization eliminate sub-optimal local minima for neural networks? An affirmative answer was given by a classical result in [59] for 1-hidden-layer wide neural networks. A few recent works have extended the setting to multi-layer neural networks, but none of them has proved every local minimum is global. Why is this result never extended to deep networks?\n  In this paper, we show that the task is impossible because the original result for 1-hidden-layer network in [59] can not hold. More specifically, we prove that for any multi-layer network with generic input data and non-lin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.01413","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-04T18:56:58Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"e96189ef96a4db40ff5d6277a94650f6685ca5e224383a0a7f7469efdfb69d37","abstract_canon_sha256":"a7c2ae76e99ba9a01a647ab297c0d9241825227f5940c1b287b1453700bba6c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:51:37.656990Z","signature_b64":"587oyM1ITpsQ6YBrerT/gC9UJFRQSo/ScBXUnDX+AwA1IrevtMhYtRIcP8Hid6qmBlhazj1wUaNSKwStLuxcBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52810aea24dafd63016026363868bc8d3fcde039759dffbb568f9a56a35cbdeb","last_reissued_at":"2026-07-05T01:51:37.656609Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:51:37.656609Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sub-Optimal Local Minima Exist for Neural Networks with Almost All Non-Linear Activations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Dawei Li, Ruoyu Sun, Tian Ding","submitted_at":"2019-11-04T18:56:58Z","abstract_excerpt":"Does over-parameterization eliminate sub-optimal local minima for neural networks? An affirmative answer was given by a classical result in [59] for 1-hidden-layer wide neural networks. A few recent works have extended the setting to multi-layer neural networks, but none of them has proved every local minimum is global. Why is this result never extended to deep networks?\n  In this paper, we show that the task is impossible because the original result for 1-hidden-layer network in [59] can not hold. More specifically, we prove that for any multi-layer network with generic input data and non-lin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.01413","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.01413/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.01413","created_at":"2026-07-05T01:51:37.656666+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.01413v3","created_at":"2026-07-05T01:51:37.656666+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.01413","created_at":"2026-07-05T01:51:37.656666+00:00"},{"alias_kind":"pith_short_12","alias_value":"KKAQV2RE3L6W","created_at":"2026-07-05T01:51:37.656666+00:00"},{"alias_kind":"pith_short_16","alias_value":"KKAQV2RE3L6WGALA","created_at":"2026-07-05T01:51:37.656666+00:00"},{"alias_kind":"pith_short_8","alias_value":"KKAQV2RE","created_at":"2026-07-05T01:51:37.656666+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14345","citing_title":"Convergence of difference inclusions via a diameter criterion","ref_index":252,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU","json":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU.json","graph_json":"https://pith.science/api/pith-number/KKAQV2RE3L6WGALAEY3DQ2F4RU/graph.json","events_json":"https://pith.science/api/pith-number/KKAQV2RE3L6WGALAEY3DQ2F4RU/events.json","paper":"https://pith.science/paper/KKAQV2RE"},"agent_actions":{"view_html":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU","download_json":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU.json","view_paper":"https://pith.science/paper/KKAQV2RE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.01413&json=true","fetch_graph":"https://pith.science/api/pith-number/KKAQV2RE3L6WGALAEY3DQ2F4RU/graph.json","fetch_events":"https://pith.science/api/pith-number/KKAQV2RE3L6WGALAEY3DQ2F4RU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU/action/storage_attestation","attest_author":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU/action/author_attestation","sign_citation":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU/action/citation_signature","submit_replication":"https://pith.science/pith/KKAQV2RE3L6WGALAEY3DQ2F4RU/action/replication_record"}},"created_at":"2026-07-05T01:51:37.656666+00:00","updated_at":"2026-07-05T01:51:37.656666+00:00"}