{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TJ7G5G7PINGQ5RHTQ4RENJGQSG","short_pith_number":"pith:TJ7G5G7P","schema_version":"1.0","canonical_sha256":"9a7e6e9bef434d0ec4f3872246a4d091b0f4738353f68d3cd8f08d14aefd887d","source":{"kind":"arxiv","id":"2308.06671","version":2},"attestation_state":"computed","paper":{"title":"Noise Balance and Stationary Distribution of Stochastic Gradient Descent","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hongchao Li, Liu Ziyin, Masahito Ueda","submitted_at":"2023-08-13T03:13:03Z","abstract_excerpt":"The stochastic gradient descent (SGD) algorithm is the algorithm we use to train neural networks. However, it remains poorly understood how the SGD navigates the highly nonlinear and degenerate loss landscape of a neural network. In this work, we show that the minibatch noise of SGD regularizes the solution towards a noise-balanced solution whenever the loss function contains a rescaling parameter symmetry. Because the difference between a simple diffusion process and SGD dynamics is the most significant when symmetries are present, our theory implies that the loss function symmetries constitu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.06671","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-08-13T03:13:03Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"93b132407101df3026f196273937c6c9e851b5b9fbe6739fe8739eeda9bcc7ac","abstract_canon_sha256":"534d5110848bb2e2c28b168a064961f33597460012fa8677f21606631adc5814"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:01.550656Z","signature_b64":"+KLh21DCKxYZC0sR5/AgRjPLk0XopwJdEHtO0LVRkazjaBKKeLPplILZWjf4IdfEqMMcRtpUrfUxKfRRwnpTDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a7e6e9bef434d0ec4f3872246a4d091b0f4738353f68d3cd8f08d14aefd887d","last_reissued_at":"2026-07-05T11:20:01.550198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:01.550198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Noise Balance and Stationary Distribution of Stochastic Gradient Descent","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Hongchao Li, Liu Ziyin, Masahito Ueda","submitted_at":"2023-08-13T03:13:03Z","abstract_excerpt":"The stochastic gradient descent (SGD) algorithm is the algorithm we use to train neural networks. However, it remains poorly understood how the SGD navigates the highly nonlinear and degenerate loss landscape of a neural network. In this work, we show that the minibatch noise of SGD regularizes the solution towards a noise-balanced solution whenever the loss function contains a rescaling parameter symmetry. Because the difference between a simple diffusion process and SGD dynamics is the most significant when symmetries are present, our theory implies that the loss function symmetries constitu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.06671","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.06671/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.06671","created_at":"2026-07-05T11:20:01.550257+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.06671v2","created_at":"2026-07-05T11:20:01.550257+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.06671","created_at":"2026-07-05T11:20:01.550257+00:00"},{"alias_kind":"pith_short_12","alias_value":"TJ7G5G7PINGQ","created_at":"2026-07-05T11:20:01.550257+00:00"},{"alias_kind":"pith_short_16","alias_value":"TJ7G5G7PINGQ5RHT","created_at":"2026-07-05T11:20:01.550257+00:00"},{"alias_kind":"pith_short_8","alias_value":"TJ7G5G7P","created_at":"2026-07-05T11:20:01.550257+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2406.09241","citing_title":"What is the long-run distribution of stochastic gradient descent? A large deviations analysis","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG","json":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG.json","graph_json":"https://pith.science/api/pith-number/TJ7G5G7PINGQ5RHTQ4RENJGQSG/graph.json","events_json":"https://pith.science/api/pith-number/TJ7G5G7PINGQ5RHTQ4RENJGQSG/events.json","paper":"https://pith.science/paper/TJ7G5G7P"},"agent_actions":{"view_html":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG","download_json":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG.json","view_paper":"https://pith.science/paper/TJ7G5G7P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.06671&json=true","fetch_graph":"https://pith.science/api/pith-number/TJ7G5G7PINGQ5RHTQ4RENJGQSG/graph.json","fetch_events":"https://pith.science/api/pith-number/TJ7G5G7PINGQ5RHTQ4RENJGQSG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG/action/storage_attestation","attest_author":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG/action/author_attestation","sign_citation":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG/action/citation_signature","submit_replication":"https://pith.science/pith/TJ7G5G7PINGQ5RHTQ4RENJGQSG/action/replication_record"}},"created_at":"2026-07-05T11:20:01.550257+00:00","updated_at":"2026-07-05T11:20:01.550257+00:00"}