{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M4TD5IKHODHP4HJIDVICHOYZFV","short_pith_number":"pith:M4TD5IKH","schema_version":"1.0","canonical_sha256":"67263ea14770cefe1d281d5023bb192d4f32c64ec3b4d0b95051b9125cde99bc","source":{"kind":"arxiv","id":"2505.23489","version":1},"attestation_state":"computed","paper":{"title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dmitry Vetrov, Ekaterina Lobacheva, Ildus Sadrtdinov, Ivan Klimov","submitted_at":"2025-05-29T14:40:24Z","abstract_excerpt":"We present a thermodynamic interpretation of the stationary behavior of stochastic gradient descent (SGD) under fixed learning rates (LRs) in neural network training. We show that SGD implicitly minimizes a free energy function $F=U-TS$, balancing training loss $U$ and the entropy of the weights distribution $S$, with temperature $T$ determined by the LR. This perspective offers a new lens on why high LRs prevent training from converging to the loss minima and how different LRs lead to stabilization at different loss levels. We empirically validate the free energy framework on both underparame"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23489","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T14:40:24Z","cross_cats_sorted":[],"title_canon_sha256":"4bedc00c5f1fa43770254958eb4b4254f0229ef4732e3cae3aebfd4bd5b7ab68","abstract_canon_sha256":"ee6a9872c0e5deecdd4c125b81c034632789f6ad645d75fcab817704986ec32d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:02.869623Z","signature_b64":"OBIn5Cp/hW7HvNOGgX51hVlHyGLgpTict6wlSgj+G/C2l/+1Pu2Q5sGDrBZnvzPEd3y+shETsylHaodBInYsCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"67263ea14770cefe1d281d5023bb192d4f32c64ec3b4d0b95051b9125cde99bc","last_reissued_at":"2026-07-05T11:12:02.869086Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:02.869086Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SGD as Free Energy Minimization: A Thermodynamic View on Neural Network Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dmitry Vetrov, Ekaterina Lobacheva, Ildus Sadrtdinov, Ivan Klimov","submitted_at":"2025-05-29T14:40:24Z","abstract_excerpt":"We present a thermodynamic interpretation of the stationary behavior of stochastic gradient descent (SGD) under fixed learning rates (LRs) in neural network training. We show that SGD implicitly minimizes a free energy function $F=U-TS$, balancing training loss $U$ and the entropy of the weights distribution $S$, with temperature $T$ determined by the LR. This perspective offers a new lens on why high LRs prevent training from converging to the loss minima and how different LRs lead to stabilization at different loss levels. We empirically validate the free energy framework on both underparame"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23489","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23489/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23489","created_at":"2026-07-05T11:12:02.869149+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23489v1","created_at":"2026-07-05T11:12:02.869149+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23489","created_at":"2026-07-05T11:12:02.869149+00:00"},{"alias_kind":"pith_short_12","alias_value":"M4TD5IKHODHP","created_at":"2026-07-05T11:12:02.869149+00:00"},{"alias_kind":"pith_short_16","alias_value":"M4TD5IKHODHP4HJI","created_at":"2026-07-05T11:12:02.869149+00:00"},{"alias_kind":"pith_short_8","alias_value":"M4TD5IKH","created_at":"2026-07-05T11:12:02.869149+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV","json":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV.json","graph_json":"https://pith.science/api/pith-number/M4TD5IKHODHP4HJIDVICHOYZFV/graph.json","events_json":"https://pith.science/api/pith-number/M4TD5IKHODHP4HJIDVICHOYZFV/events.json","paper":"https://pith.science/paper/M4TD5IKH"},"agent_actions":{"view_html":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV","download_json":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV.json","view_paper":"https://pith.science/paper/M4TD5IKH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23489&json=true","fetch_graph":"https://pith.science/api/pith-number/M4TD5IKHODHP4HJIDVICHOYZFV/graph.json","fetch_events":"https://pith.science/api/pith-number/M4TD5IKHODHP4HJIDVICHOYZFV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV/action/storage_attestation","attest_author":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV/action/author_attestation","sign_citation":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV/action/citation_signature","submit_replication":"https://pith.science/pith/M4TD5IKHODHP4HJIDVICHOYZFV/action/replication_record"}},"created_at":"2026-07-05T11:12:02.869149+00:00","updated_at":"2026-07-05T11:12:02.869149+00:00"}