{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I6DYAHCRPXIDWQD2YN7B6XO2OP","short_pith_number":"pith:I6DYAHCR","schema_version":"1.0","canonical_sha256":"4787801c517dd03b407ac37e1f5dda73ff17dfd8f5338e24dfc28964e629a4e1","source":{"kind":"arxiv","id":"2405.13698","version":3},"attestation_state":"computed","paper":{"title":"How to set AdamW's weight decay as you scale model and dataset size","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Laurence Aitchison, Xi Wang","submitted_at":"2024-05-22T14:43:02Z","abstract_excerpt":"The scaling of the optimal AdamW weight decay hyperparameter with model and dataset size is critical as we seek to build larger models, but is poorly understood. We show that weights learned by AdamW can be understood as an exponential moving average (EMA) of recent updates. This gives critical insights for how to set the weight decay in AdamW, and how the weight decay should scale with model and dataset size. In particular, the key hyperparameter for an exponential moving average is the EMA timescale. Intuitively, the EMA timescale can be understood as the number of recent iterations the EMA "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13698","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-22T14:43:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"60f5a6dba065aad38c2755b3ffa9a7b638871bd98da2ae4e67cbbb3fbbda9809","abstract_canon_sha256":"b198faf1cd92e511d937f9847a533da60d64c338a64f7d2ca6c8d42e4cc9eee9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:09.862334Z","signature_b64":"4vLZPjGd5aJr4Luiwusc3Gq1R49VkM8/8OIa/ZIU1rwnugVHByXZUKK0u+WopGkh5hxFKqXYL0gBPhwqVeF/DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4787801c517dd03b407ac37e1f5dda73ff17dfd8f5338e24dfc28964e629a4e1","last_reissued_at":"2026-07-05T11:13:09.861774Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:09.861774Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to set AdamW's weight decay as you scale model and dataset size","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Laurence Aitchison, Xi Wang","submitted_at":"2024-05-22T14:43:02Z","abstract_excerpt":"The scaling of the optimal AdamW weight decay hyperparameter with model and dataset size is critical as we seek to build larger models, but is poorly understood. We show that weights learned by AdamW can be understood as an exponential moving average (EMA) of recent updates. This gives critical insights for how to set the weight decay in AdamW, and how the weight decay should scale with model and dataset size. In particular, the key hyperparameter for an exponential moving average is the EMA timescale. Intuitively, the EMA timescale can be understood as the number of recent iterations the EMA "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13698","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13698/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13698","created_at":"2026-07-05T11:13:09.861838+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13698v3","created_at":"2026-07-05T11:13:09.861838+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13698","created_at":"2026-07-05T11:13:09.861838+00:00"},{"alias_kind":"pith_short_12","alias_value":"I6DYAHCRPXID","created_at":"2026-07-05T11:13:09.861838+00:00"},{"alias_kind":"pith_short_16","alias_value":"I6DYAHCRPXIDWQD2","created_at":"2026-07-05T11:13:09.861838+00:00"},{"alias_kind":"pith_short_8","alias_value":"I6DYAHCR","created_at":"2026-07-05T11:13:09.861838+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18898","citing_title":"A Two-Parameter Weibull Framework for Diagnosing Transformer Weight Distributions","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15290","citing_title":"GQA-{\\mu}P: The maximal parameterization update for grouped query attention","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28743","citing_title":"Rethinking Language Model Scaling under Transferable Hypersphere Optimization","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP","json":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP.json","graph_json":"https://pith.science/api/pith-number/I6DYAHCRPXIDWQD2YN7B6XO2OP/graph.json","events_json":"https://pith.science/api/pith-number/I6DYAHCRPXIDWQD2YN7B6XO2OP/events.json","paper":"https://pith.science/paper/I6DYAHCR"},"agent_actions":{"view_html":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP","download_json":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP.json","view_paper":"https://pith.science/paper/I6DYAHCR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13698&json=true","fetch_graph":"https://pith.science/api/pith-number/I6DYAHCRPXIDWQD2YN7B6XO2OP/graph.json","fetch_events":"https://pith.science/api/pith-number/I6DYAHCRPXIDWQD2YN7B6XO2OP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP/action/storage_attestation","attest_author":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP/action/author_attestation","sign_citation":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP/action/citation_signature","submit_replication":"https://pith.science/pith/I6DYAHCRPXIDWQD2YN7B6XO2OP/action/replication_record"}},"created_at":"2026-07-05T11:13:09.861838+00:00","updated_at":"2026-07-05T11:13:09.861838+00:00"}