{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:BLZ3FREMKJPM6RAOB2533O7SCF","short_pith_number":"pith:BLZ3FREM","schema_version":"1.0","canonical_sha256":"0af3b2c48c525ecf440e0ebbbdbbf211657117c1ec37358325dcd3338a1475bf","source":{"kind":"arxiv","id":"1810.12281","version":1},"attestation_state":"computed","paper":{"title":"Three Mechanisms of Weight Decay Regularization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bowen Xu, Chaoqi Wang, Guodong Zhang, Roger Grosse","submitted_at":"2018-10-29T17:51:25Z","abstract_excerpt":"Weight decay is one of the standard tricks in the neural network toolbox, but the reasons for its regularization effect are poorly understood, and recent results have cast doubt on the traditional interpretation in terms of $L_2$ regularization. Literal weight decay has been shown to outperform $L_2$ regularization for optimizers for which they differ. We empirically investigate weight decay for three optimization algorithms (SGD, Adam, and K-FAC) and a variety of network architectures. We identify three distinct mechanisms by which weight decay exerts a regularization effect, depending on the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1810.12281","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-10-29T17:51:25Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"ec479b64476ee39129e5a4bdc02c1315af06784d4ef1c3a015fe5ed925ee0ec5","abstract_canon_sha256":"d967324fd296dedb9c0c38323136c2c291635c6c7677e7ce61dc6d582afb97ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:02:03.731029Z","signature_b64":"WsuVzMpGzyIVqesJ3P2UZ07W2ijoco5fGa0VmYc+cgEdvKDeGRbu8JO83nPuvdKO7APHHsTKcCy+hWEvRzkRBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0af3b2c48c525ecf440e0ebbbdbbf211657117c1ec37358325dcd3338a1475bf","last_reissued_at":"2026-05-18T00:02:03.730361Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:02:03.730361Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Three Mechanisms of Weight Decay Regularization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bowen Xu, Chaoqi Wang, Guodong Zhang, Roger Grosse","submitted_at":"2018-10-29T17:51:25Z","abstract_excerpt":"Weight decay is one of the standard tricks in the neural network toolbox, but the reasons for its regularization effect are poorly understood, and recent results have cast doubt on the traditional interpretation in terms of $L_2$ regularization. Literal weight decay has been shown to outperform $L_2$ regularization for optimizers for which they differ. We empirically investigate weight decay for three optimization algorithms (SGD, Adam, and K-FAC) and a variety of network architectures. We identify three distinct mechanisms by which weight decay exerts a regularization effect, depending on the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1810.12281","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1810.12281","created_at":"2026-05-18T00:02:03.730479+00:00"},{"alias_kind":"arxiv_version","alias_value":"1810.12281v1","created_at":"2026-05-18T00:02:03.730479+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1810.12281","created_at":"2026-05-18T00:02:03.730479+00:00"},{"alias_kind":"pith_short_12","alias_value":"BLZ3FREMKJPM","created_at":"2026-05-18T12:32:16.446611+00:00"},{"alias_kind":"pith_short_16","alias_value":"BLZ3FREMKJPM6RAO","created_at":"2026-05-18T12:32:16.446611+00:00"},{"alias_kind":"pith_short_8","alias_value":"BLZ3FREM","created_at":"2026-05-18T12:32:16.446611+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2602.17219","citing_title":"Vibrational infrared and Raman spectra of the methanol molecule with equivariant neural-network property surfaces","ref_index":102,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04418","citing_title":"Demystifying Manifold Constraints in LLM Pre-training","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03602","citing_title":"Dante: An Open Source Model Pre-Training and Fine-Tuning Tool for the Dafne Federated Framework for Medical Image Segmentation","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF","json":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF.json","graph_json":"https://pith.science/api/pith-number/BLZ3FREMKJPM6RAOB2533O7SCF/graph.json","events_json":"https://pith.science/api/pith-number/BLZ3FREMKJPM6RAOB2533O7SCF/events.json","paper":"https://pith.science/paper/BLZ3FREM"},"agent_actions":{"view_html":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF","download_json":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF.json","view_paper":"https://pith.science/paper/BLZ3FREM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1810.12281&json=true","fetch_graph":"https://pith.science/api/pith-number/BLZ3FREMKJPM6RAOB2533O7SCF/graph.json","fetch_events":"https://pith.science/api/pith-number/BLZ3FREMKJPM6RAOB2533O7SCF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF/action/storage_attestation","attest_author":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF/action/author_attestation","sign_citation":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF/action/citation_signature","submit_replication":"https://pith.science/pith/BLZ3FREMKJPM6RAOB2533O7SCF/action/replication_record"}},"created_at":"2026-05-18T00:02:03.730479+00:00","updated_at":"2026-05-18T00:02:03.730479+00:00"}