{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:CGEVZ6NK52TKQ2FCSHNG7K5U3F","short_pith_number":"pith:CGEVZ6NK","schema_version":"1.0","canonical_sha256":"11895cf9aaeea6a868a291da6fabb4d97134559b8fdae5825c8e7e4e7a8776d4","source":{"kind":"arxiv","id":"2012.03636","version":4},"attestation_state":"computed","paper":{"title":"Noise and Fluctuation of Finite Learning Rate Stochastic Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Kangqiao Liu, Liu Ziyin, Masahito Ueda","submitted_at":"2020-12-07T12:31:43Z","abstract_excerpt":"In the vanishing learning rate regime, stochastic gradient descent (SGD) is now relatively well understood. In this work, we propose to study the basic properties of SGD and its variants in the non-vanishing learning rate regime. The focus is on deriving exactly solvable results and discussing their implications. The main contributions of this work are to derive the stationary distribution for discrete-time SGD in a quadratic loss function with and without momentum; in particular, one implication of our result is that the fluctuation caused by discrete-time dynamics takes a distorted shape and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.03636","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2020-12-07T12:31:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1c55c39328dda594c53897020b83bc96e541c1213a7cec2c63870fdca1e5e573","abstract_canon_sha256":"02b8e705b76837c58ea81f42ed4165d9e9913b862fef425239bd7fb909af0670"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:20.227002Z","signature_b64":"zuKV3vSGtBjNFKonH0xfNLynTq2kW4fM65dAfotel8wjY6P20H1HcALd+2Zgsj1+iD0Z1crgPcukth2u7vk0Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11895cf9aaeea6a868a291da6fabb4d97134559b8fdae5825c8e7e4e7a8776d4","last_reissued_at":"2026-07-05T02:48:20.226564Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:20.226564Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Noise and Fluctuation of Finite Learning Rate Stochastic Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Kangqiao Liu, Liu Ziyin, Masahito Ueda","submitted_at":"2020-12-07T12:31:43Z","abstract_excerpt":"In the vanishing learning rate regime, stochastic gradient descent (SGD) is now relatively well understood. In this work, we propose to study the basic properties of SGD and its variants in the non-vanishing learning rate regime. The focus is on deriving exactly solvable results and discussing their implications. The main contributions of this work are to derive the stationary distribution for discrete-time SGD in a quadratic loss function with and without momentum; in particular, one implication of our result is that the fluctuation caused by discrete-time dynamics takes a distorted shape and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.03636","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.03636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.03636","created_at":"2026-07-05T02:48:20.226617+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.03636v4","created_at":"2026-07-05T02:48:20.226617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.03636","created_at":"2026-07-05T02:48:20.226617+00:00"},{"alias_kind":"pith_short_12","alias_value":"CGEVZ6NK52TK","created_at":"2026-07-05T02:48:20.226617+00:00"},{"alias_kind":"pith_short_16","alias_value":"CGEVZ6NK52TKQ2FC","created_at":"2026-07-05T02:48:20.226617+00:00"},{"alias_kind":"pith_short_8","alias_value":"CGEVZ6NK","created_at":"2026-07-05T02:48:20.226617+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.23489","citing_title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F","json":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F.json","graph_json":"https://pith.science/api/pith-number/CGEVZ6NK52TKQ2FCSHNG7K5U3F/graph.json","events_json":"https://pith.science/api/pith-number/CGEVZ6NK52TKQ2FCSHNG7K5U3F/events.json","paper":"https://pith.science/paper/CGEVZ6NK"},"agent_actions":{"view_html":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F","download_json":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F.json","view_paper":"https://pith.science/paper/CGEVZ6NK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.03636&json=true","fetch_graph":"https://pith.science/api/pith-number/CGEVZ6NK52TKQ2FCSHNG7K5U3F/graph.json","fetch_events":"https://pith.science/api/pith-number/CGEVZ6NK52TKQ2FCSHNG7K5U3F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F/action/storage_attestation","attest_author":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F/action/author_attestation","sign_citation":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F/action/citation_signature","submit_replication":"https://pith.science/pith/CGEVZ6NK52TKQ2FCSHNG7K5U3F/action/replication_record"}},"created_at":"2026-07-05T02:48:20.226617+00:00","updated_at":"2026-07-05T02:48:20.226617+00:00"}