{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:LWNYDF6UWY24QXCE7TJMJJZU4A","short_pith_number":"pith:LWNYDF6U","schema_version":"1.0","canonical_sha256":"5d9b8197d4b635c85c44fcd2c4a734e00455e9e65ebaca2ba4c05c04527dc900","source":{"kind":"arxiv","id":"1910.00195","version":2},"attestation_state":"computed","paper":{"title":"How noise affects the Hessian spectrum in overparameterized neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"David J Schwab, Mingwei Wei","submitted_at":"2019-10-01T04:13:27Z","abstract_excerpt":"Stochastic gradient descent (SGD) forms the core optimization method for deep neural networks. While some theoretical progress has been made, it still remains unclear why SGD leads the learning dynamics in overparameterized networks to solutions that generalize well. Here we show that for overparameterized networks with a degenerate valley in their loss landscape, SGD on average decreases the trace of the Hessian of the loss. We also generalize this result to other noise structures and show that isotropic noise in the non-degenerate subspace of the Hessian decreases its determinant. In additio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.00195","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-01T04:13:27Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"feeae0c056fcb0b398b267d6a333bdaf1bf06766536b1bf8b51b09c6c9e25f5a","abstract_canon_sha256":"77be5d5608f482677d6d51a1b12461c2ed039a4032bc791ba8e54953fc19c7f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:15:41.112986Z","signature_b64":"7dv8jXt+usJcqGl4D/MtTUS4NwhoDfgXQeZiBL4Tm4VpGmNtHj6U07CTskX513jIYvYqZQ9FUuyzU8iaEX6pAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d9b8197d4b635c85c44fcd2c4a734e00455e9e65ebaca2ba4c05c04527dc900","last_reissued_at":"2026-07-05T00:15:41.112548Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:15:41.112548Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How noise affects the Hessian spectrum in overparameterized neural networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"David J Schwab, Mingwei Wei","submitted_at":"2019-10-01T04:13:27Z","abstract_excerpt":"Stochastic gradient descent (SGD) forms the core optimization method for deep neural networks. While some theoretical progress has been made, it still remains unclear why SGD leads the learning dynamics in overparameterized networks to solutions that generalize well. Here we show that for overparameterized networks with a degenerate valley in their loss landscape, SGD on average decreases the trace of the Hessian of the loss. We also generalize this result to other noise structures and show that isotropic noise in the non-degenerate subspace of the Hessian decreases its determinant. In additio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.00195","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.00195/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.00195","created_at":"2026-07-05T00:15:41.112603+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.00195v2","created_at":"2026-07-05T00:15:41.112603+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.00195","created_at":"2026-07-05T00:15:41.112603+00:00"},{"alias_kind":"pith_short_12","alias_value":"LWNYDF6UWY24","created_at":"2026-07-05T00:15:41.112603+00:00"},{"alias_kind":"pith_short_16","alias_value":"LWNYDF6UWY24QXCE","created_at":"2026-07-05T00:15:41.112603+00:00"},{"alias_kind":"pith_short_8","alias_value":"LWNYDF6U","created_at":"2026-07-05T00:15:41.112603+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28662","citing_title":"Closed-Form Steepest Descent Direction toward Flat Minima: Reducing Upper Bounds on the Loss Hessian Eigenspectrum in Neural Networks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10202","citing_title":"Wolkowicz-Styan Upper Bound on the Hessian Eigenspectrum for Cross-Entropy Loss in Nonlinear Smooth Neural Networks","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A","json":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A.json","graph_json":"https://pith.science/api/pith-number/LWNYDF6UWY24QXCE7TJMJJZU4A/graph.json","events_json":"https://pith.science/api/pith-number/LWNYDF6UWY24QXCE7TJMJJZU4A/events.json","paper":"https://pith.science/paper/LWNYDF6U"},"agent_actions":{"view_html":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A","download_json":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A.json","view_paper":"https://pith.science/paper/LWNYDF6U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.00195&json=true","fetch_graph":"https://pith.science/api/pith-number/LWNYDF6UWY24QXCE7TJMJJZU4A/graph.json","fetch_events":"https://pith.science/api/pith-number/LWNYDF6UWY24QXCE7TJMJJZU4A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A/action/storage_attestation","attest_author":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A/action/author_attestation","sign_citation":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A/action/citation_signature","submit_replication":"https://pith.science/pith/LWNYDF6UWY24QXCE7TJMJJZU4A/action/replication_record"}},"created_at":"2026-07-05T00:15:41.112603+00:00","updated_at":"2026-07-05T00:15:41.112603+00:00"}