{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OJE2ITAO7N2OTQ3DQFVBXPXKZY","short_pith_number":"pith:OJE2ITAO","schema_version":"1.0","canonical_sha256":"7249a44c0efb74e9c363816a1bbeeace366454014400dc77059fbb76b123d98f","source":{"kind":"arxiv","id":"2305.16891","version":2},"attestation_state":"computed","paper":{"title":"Generalization Guarantees of Gradient Descent for Multi-Layer Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ding-Xuan Zhou, Di Wang, Puyu Wang, Yiming Ying, Yunwen Lei","submitted_at":"2023-05-26T12:51:38Z","abstract_excerpt":"Recently, significant progress has been made in understanding the generalization of neural networks (NNs) trained by gradient descent (GD) using the algorithmic stability approach. However, most of the existing research has focused on one-hidden-layer NNs and has not addressed the impact of different network scaling parameters. In this paper, we greatly extend the previous work \\cite{lei2022stability,richards2021stability} by conducting a comprehensive stability and generalization analysis of GD for multi-layer NNs. For two-layer NNs, our results are established under general network scaling p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16891","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-05-26T12:51:38Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"ef01100688e2c84c920b727e5ff2f95de643f62bd221a5d15a05154b6eb3fc6b","abstract_canon_sha256":"2ee545572f03acd3e2ef8df4fdf4184c2d088d52f6fa0c0f0528384ef599e72d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:40:29.127792Z","signature_b64":"LEtEMJi7q5COOWg7utYslb9td3mY10ZmaH/2OcM9lju9qWQBQgoMZTb2pA/PVmZgJxS46xnUKK9a9lw0C6vKBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7249a44c0efb74e9c363816a1bbeeace366454014400dc77059fbb76b123d98f","last_reissued_at":"2026-07-05T11:40:29.127291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:40:29.127291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalization Guarantees of Gradient Descent for Multi-Layer Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ding-Xuan Zhou, Di Wang, Puyu Wang, Yiming Ying, Yunwen Lei","submitted_at":"2023-05-26T12:51:38Z","abstract_excerpt":"Recently, significant progress has been made in understanding the generalization of neural networks (NNs) trained by gradient descent (GD) using the algorithmic stability approach. However, most of the existing research has focused on one-hidden-layer NNs and has not addressed the impact of different network scaling parameters. In this paper, we greatly extend the previous work \\cite{lei2022stability,richards2021stability} by conducting a comprehensive stability and generalization analysis of GD for multi-layer NNs. For two-layer NNs, our results are established under general network scaling p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16891","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16891/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16891","created_at":"2026-07-05T11:40:29.127363+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16891v2","created_at":"2026-07-05T11:40:29.127363+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16891","created_at":"2026-07-05T11:40:29.127363+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJE2ITAO7N2O","created_at":"2026-07-05T11:40:29.127363+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJE2ITAO7N2OTQ3D","created_at":"2026-07-05T11:40:29.127363+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJE2ITAO","created_at":"2026-07-05T11:40:29.127363+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.08282","citing_title":"How Does the Smoothness Approximation Method Facilitate Generalization for Federated Adversarial Learning?","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY","json":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY.json","graph_json":"https://pith.science/api/pith-number/OJE2ITAO7N2OTQ3DQFVBXPXKZY/graph.json","events_json":"https://pith.science/api/pith-number/OJE2ITAO7N2OTQ3DQFVBXPXKZY/events.json","paper":"https://pith.science/paper/OJE2ITAO"},"agent_actions":{"view_html":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY","download_json":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY.json","view_paper":"https://pith.science/paper/OJE2ITAO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16891&json=true","fetch_graph":"https://pith.science/api/pith-number/OJE2ITAO7N2OTQ3DQFVBXPXKZY/graph.json","fetch_events":"https://pith.science/api/pith-number/OJE2ITAO7N2OTQ3DQFVBXPXKZY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY/action/storage_attestation","attest_author":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY/action/author_attestation","sign_citation":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY/action/citation_signature","submit_replication":"https://pith.science/pith/OJE2ITAO7N2OTQ3DQFVBXPXKZY/action/replication_record"}},"created_at":"2026-07-05T11:40:29.127363+00:00","updated_at":"2026-07-05T11:40:29.127363+00:00"}