{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:SHOBHRPUCQ6U63KKNCH7PUXLHN","short_pith_number":"pith:SHOBHRPU","schema_version":"1.0","canonical_sha256":"91dc13c5f4143d4f6d4a688ff7d2eb3b4ee3396915a9b053b24ad977e48b455d","source":{"kind":"arxiv","id":"1904.04326","version":2},"attestation_state":"computed","paper":{"title":"A Comparative Analysis of the Optimization and Generalization Property of Two-layer Neural Network and Random Feature Models Under Gradient Descent Dynamics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chao Ma, Lei Wu, Weinan E","submitted_at":"2019-04-08T19:43:09Z","abstract_excerpt":"A fairly comprehensive analysis is presented for the gradient descent dynamics for training two-layer neural network models in the situation when the parameters in both layers are updated. General initialization schemes as well as general regimes for the network width and training data size are considered. In the over-parametrized regime, it is shown that gradient descent dynamics can achieve zero training loss exponentially fast regardless of the quality of the labels. In addition, it is proved that throughout the training process the functions represented by the neural network model are unif"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.04326","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-04-08T19:43:09Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"10e9605746eac59f60006db50cd4bf3cdc983c734337995113edbb0341641034","abstract_canon_sha256":"b80c2ab9db9b53a726de46b0c79984d302cf41c012ce11ca3095d8b5a4d073a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:43:54.032851Z","signature_b64":"Mel3gtmxaLcv4/nRbVNOhKaSnatdP9xQfEymLEUSGlwnmLbz7Lj6aFSBJHHbAmLDiTnb6ZvSAHMOGpUx6FF6Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"91dc13c5f4143d4f6d4a688ff7d2eb3b4ee3396915a9b053b24ad977e48b455d","last_reissued_at":"2026-07-05T00:43:54.032421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:43:54.032421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comparative Analysis of the Optimization and Generalization Property of Two-layer Neural Network and Random Feature Models Under Gradient Descent Dynamics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chao Ma, Lei Wu, Weinan E","submitted_at":"2019-04-08T19:43:09Z","abstract_excerpt":"A fairly comprehensive analysis is presented for the gradient descent dynamics for training two-layer neural network models in the situation when the parameters in both layers are updated. General initialization schemes as well as general regimes for the network width and training data size are considered. In the over-parametrized regime, it is shown that gradient descent dynamics can achieve zero training loss exponentially fast regardless of the quality of the labels. In addition, it is proved that throughout the training process the functions represented by the neural network model are unif"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.04326","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.04326/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.04326","created_at":"2026-07-05T00:43:54.032486+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.04326v2","created_at":"2026-07-05T00:43:54.032486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.04326","created_at":"2026-07-05T00:43:54.032486+00:00"},{"alias_kind":"pith_short_12","alias_value":"SHOBHRPUCQ6U","created_at":"2026-07-05T00:43:54.032486+00:00"},{"alias_kind":"pith_short_16","alias_value":"SHOBHRPUCQ6U63KK","created_at":"2026-07-05T00:43:54.032486+00:00"},{"alias_kind":"pith_short_8","alias_value":"SHOBHRPU","created_at":"2026-07-05T00:43:54.032486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27142","citing_title":"Estimation of High Dimensional Bounded Discrete Graphical Models via Regularized Generalized Score Matching","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10089","citing_title":"A Theory on Flow Matching with Neural Networks","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"1906.08654","citing_title":"ID3 Learns Juntas for Smoothed Product Distributions","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2401.01335","citing_title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN","json":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN.json","graph_json":"https://pith.science/api/pith-number/SHOBHRPUCQ6U63KKNCH7PUXLHN/graph.json","events_json":"https://pith.science/api/pith-number/SHOBHRPUCQ6U63KKNCH7PUXLHN/events.json","paper":"https://pith.science/paper/SHOBHRPU"},"agent_actions":{"view_html":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN","download_json":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN.json","view_paper":"https://pith.science/paper/SHOBHRPU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.04326&json=true","fetch_graph":"https://pith.science/api/pith-number/SHOBHRPUCQ6U63KKNCH7PUXLHN/graph.json","fetch_events":"https://pith.science/api/pith-number/SHOBHRPUCQ6U63KKNCH7PUXLHN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN/action/storage_attestation","attest_author":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN/action/author_attestation","sign_citation":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN/action/citation_signature","submit_replication":"https://pith.science/pith/SHOBHRPUCQ6U63KKNCH7PUXLHN/action/replication_record"}},"created_at":"2026-07-05T00:43:54.032486+00:00","updated_at":"2026-07-05T00:43:54.032486+00:00"}