{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:MYC6432EGFYRMMIS77JQIK63ON","short_pith_number":"pith:MYC6432E","schema_version":"1.0","canonical_sha256":"6605ee6f443171163112ffd3042bdb734800c911113a451755f3787c4a862fd0","source":{"kind":"arxiv","id":"1905.09870","version":3},"attestation_state":"computed","paper":{"title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Atsushi Nitanda, Geoffrey Chinot, Taiji Suzuki","submitted_at":"2019-05-23T18:57:35Z","abstract_excerpt":"Recently, several studies have proven the global convergence and generalization abilities of the gradient descent method for two-layer ReLU networks. Most studies especially focused on the regression problems with the squared loss function, except for a few, and the importance of the positivity of the neural tangent kernel has been pointed out. On the other hand, the performance of gradient descent on classification problems using the logistic loss function has not been well studied, and further investigation of this problem structure is possible. In this work, we demonstrate that the separabi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.09870","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2019-05-23T18:57:35Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a746582c2fca89815bfd91855d28192fc518c65805505d12dd1bcfc6b6f0b8be","abstract_canon_sha256":"1f1d15d8e063e540f6dda9ed43dfafcca6c3623365e6df2159a53806d7f196e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:48:59.170873Z","signature_b64":"WONYRiUNsb6BI0CkGVb80H9XA3dOM9Kc10aD6tVN9PwdugtgeIQOlh9awHNdHkq7R8OnnH4tl8Dfn9nj1voABw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6605ee6f443171163112ffd3042bdb734800c911113a451755f3787c4a862fd0","last_reissued_at":"2026-07-05T00:48:59.170405Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:48:59.170405Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Gradient Descent can Learn Less Over-parameterized Two-layer Neural Networks on Classification Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Atsushi Nitanda, Geoffrey Chinot, Taiji Suzuki","submitted_at":"2019-05-23T18:57:35Z","abstract_excerpt":"Recently, several studies have proven the global convergence and generalization abilities of the gradient descent method for two-layer ReLU networks. Most studies especially focused on the regression problems with the squared loss function, except for a few, and the importance of the positivity of the neural tangent kernel has been pointed out. On the other hand, the performance of gradient descent on classification problems using the logistic loss function has not been well studied, and further investigation of this problem structure is possible. In this work, we demonstrate that the separabi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.09870","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1905.09870/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.09870","created_at":"2026-07-05T00:48:59.170461+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.09870v3","created_at":"2026-07-05T00:48:59.170461+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.09870","created_at":"2026-07-05T00:48:59.170461+00:00"},{"alias_kind":"pith_short_12","alias_value":"MYC6432EGFYR","created_at":"2026-07-05T00:48:59.170461+00:00"},{"alias_kind":"pith_short_16","alias_value":"MYC6432EGFYRMMIS","created_at":"2026-07-05T00:48:59.170461+00:00"},{"alias_kind":"pith_short_8","alias_value":"MYC6432E","created_at":"2026-07-05T00:48:59.170461+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27142","citing_title":"Estimation of High Dimensional Bounded Discrete Graphical Models via Regularized Generalized Score Matching","ref_index":256,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26749","citing_title":"Structure Before Collapse: Transient semantic geometry in next-token prediction","ref_index":272,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10089","citing_title":"A Theory on Flow Matching with Neural Networks","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06772","citing_title":"Minimax-Optimal Generalization Bounds for Smooth Deep Neural Networks Trained by (Stochastic) Gradient Descent","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06764","citing_title":"Optimal Rates for Generalization of Gradient Descent Methods with Deep Neural Networks","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22409","citing_title":"Optimization, Generalization and Differential Privacy Bounds for Gradient Descent on Kolmogorov-Arnold Networks","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2401.01335","citing_title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12648","citing_title":"Population Risk Bounds for Kolmogorov-Arnold Networks Trained by DP-SGD with Correlated Noise","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON","json":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON.json","graph_json":"https://pith.science/api/pith-number/MYC6432EGFYRMMIS77JQIK63ON/graph.json","events_json":"https://pith.science/api/pith-number/MYC6432EGFYRMMIS77JQIK63ON/events.json","paper":"https://pith.science/paper/MYC6432E"},"agent_actions":{"view_html":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON","download_json":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON.json","view_paper":"https://pith.science/paper/MYC6432E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.09870&json=true","fetch_graph":"https://pith.science/api/pith-number/MYC6432EGFYRMMIS77JQIK63ON/graph.json","fetch_events":"https://pith.science/api/pith-number/MYC6432EGFYRMMIS77JQIK63ON/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON/action/storage_attestation","attest_author":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON/action/author_attestation","sign_citation":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON/action/citation_signature","submit_replication":"https://pith.science/pith/MYC6432EGFYRMMIS77JQIK63ON/action/replication_record"}},"created_at":"2026-07-05T00:48:59.170461+00:00","updated_at":"2026-07-05T00:48:59.170461+00:00"}