{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FIIRWZ7EWWL7SRPNTY47XVFBFX","short_pith_number":"pith:FIIRWZ7E","schema_version":"1.0","canonical_sha256":"2a111b67e4b597f945ed9e39fbd4a12df5c3b828c153bf02cc75a110080e1516","source":{"kind":"arxiv","id":"2506.02285","version":2},"attestation_state":"computed","paper":{"title":"Why Gradients Rapidly Increase Near the End of Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaron Defazio","submitted_at":"2025-06-02T21:51:04Z","abstract_excerpt":"During long-duration Large Language Model (LLM) training runs the gradient norm increases rapidly near the end of training. In this short note, we show that this increase is due to an unintended interaction between weight decay, normalization layers, and the learning rate schedule. We propose a simple correction that fixes this behavior while also resulting in lower loss values throughout training."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02285","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-02T21:51:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"44be545d3448b67b0ada1d4e688223b7525f3da4d331bde4f6dbfa23a1e1008c","abstract_canon_sha256":"5cf9588528c7dbb96a6ce86851215039ae5eeb461456adad71c63e57936e042d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:43.416001Z","signature_b64":"FRnOmccteZeFg2zZbobA6Du4HTM7Hp3jpBz7WbF8Yl6WSvBHcF4yWxHSmiI8HcZXITV3GxX9l8qXDu6sHgpwCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a111b67e4b597f945ed9e39fbd4a12df5c3b828c153bf02cc75a110080e1516","last_reissued_at":"2026-07-05T11:18:43.415559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:43.415559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Gradients Rapidly Increase Near the End of Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaron Defazio","submitted_at":"2025-06-02T21:51:04Z","abstract_excerpt":"During long-duration Large Language Model (LLM) training runs the gradient norm increases rapidly near the end of training. In this short note, we show that this increase is due to an unintended interaction between weight decay, normalization layers, and the learning rate schedule. We propose a simple correction that fixes this behavior while also resulting in lower loss values throughout training."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02285","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02285/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02285","created_at":"2026-07-05T11:18:43.415614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02285v2","created_at":"2026-07-05T11:18:43.415614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02285","created_at":"2026-07-05T11:18:43.415614+00:00"},{"alias_kind":"pith_short_12","alias_value":"FIIRWZ7EWWL7","created_at":"2026-07-05T11:18:43.415614+00:00"},{"alias_kind":"pith_short_16","alias_value":"FIIRWZ7EWWL7SRPN","created_at":"2026-07-05T11:18:43.415614+00:00"},{"alias_kind":"pith_short_8","alias_value":"FIIRWZ7E","created_at":"2026-07-05T11:18:43.415614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23061","citing_title":"Anytime Training with Schedule-Free Spectral Optimization","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28743","citing_title":"Rethinking Language Model Scaling under Transferable Hypersphere Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06654","citing_title":"Optimizer-Model Consistency: Full Finetuning with the Same Optimizer as Pretraining Forgets Less","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04418","citing_title":"Demystifying Manifold Constraints in LLM Pre-training","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13483","citing_title":"Broximal Alignment for Global Non-Convex Optimization","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX","json":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX.json","graph_json":"https://pith.science/api/pith-number/FIIRWZ7EWWL7SRPNTY47XVFBFX/graph.json","events_json":"https://pith.science/api/pith-number/FIIRWZ7EWWL7SRPNTY47XVFBFX/events.json","paper":"https://pith.science/paper/FIIRWZ7E"},"agent_actions":{"view_html":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX","download_json":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX.json","view_paper":"https://pith.science/paper/FIIRWZ7E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02285&json=true","fetch_graph":"https://pith.science/api/pith-number/FIIRWZ7EWWL7SRPNTY47XVFBFX/graph.json","fetch_events":"https://pith.science/api/pith-number/FIIRWZ7EWWL7SRPNTY47XVFBFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX/action/storage_attestation","attest_author":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX/action/author_attestation","sign_citation":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX/action/citation_signature","submit_replication":"https://pith.science/pith/FIIRWZ7EWWL7SRPNTY47XVFBFX/action/replication_record"}},"created_at":"2026-07-05T11:18:43.415614+00:00","updated_at":"2026-07-05T11:18:43.415614+00:00"}