{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XPHPVTNMRTQUB7QIDK4N53MI4G","short_pith_number":"pith:XPHPVTNM","schema_version":"1.0","canonical_sha256":"bbcefacdac8ce140fe081ab8deed88e18851306c0a8c79dc7f496c5c6e229a2f","source":{"kind":"arxiv","id":"2003.02218","version":1},"attestation_state":"computed","paper":{"title":"The large learning rate phase of deep learning: the catapult mechanism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Aitor Lewkowycz, Ethan Dyer, Guy Gur-Ari, Jascha Sohl-Dickstein, Yasaman Bahri","submitted_at":"2020-03-04T17:52:48Z","abstract_excerpt":"The choice of initial learning rate can have a profound effect on the performance of deep networks. We present a class of neural networks with solvable training dynamics, and confirm their predictions empirically in practical deep learning settings. The networks exhibit sharply distinct behaviors at small and large learning rates. The two regimes are separated by a phase transition. In the small learning rate phase, training can be understood using the existing theory of infinitely wide neural networks. At large learning rates the model captures qualitatively distinct phenomena, including the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.02218","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2020-03-04T17:52:48Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"43c27608e8e0be6224be8e550a4472b8d31ca4eb0fdf7b3b68deebe9e9bbb55c","abstract_canon_sha256":"6768c0bc8a5e018209c1cab584cdfb1ddee2db971b490e82d1235977ac57bcaf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:45:47.919066Z","signature_b64":"WdANNRDcb9fcdHq1Ij+fXubs/OaNd8ceciYcjdE3Rz4KOnfW4LYfXZf2S9aslGaLY2MlxGKRTYIs++PLU+Z8Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bbcefacdac8ce140fe081ab8deed88e18851306c0a8c79dc7f496c5c6e229a2f","last_reissued_at":"2026-07-05T00:45:47.918590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:45:47.918590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The large learning rate phase of deep learning: the catapult mechanism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Aitor Lewkowycz, Ethan Dyer, Guy Gur-Ari, Jascha Sohl-Dickstein, Yasaman Bahri","submitted_at":"2020-03-04T17:52:48Z","abstract_excerpt":"The choice of initial learning rate can have a profound effect on the performance of deep networks. We present a class of neural networks with solvable training dynamics, and confirm their predictions empirically in practical deep learning settings. The networks exhibit sharply distinct behaviors at small and large learning rates. The two regimes are separated by a phase transition. In the small learning rate phase, training can be understood using the existing theory of infinitely wide neural networks. At large learning rates the model captures qualitatively distinct phenomena, including the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.02218","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.02218/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.02218","created_at":"2026-07-05T00:45:47.918652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.02218v1","created_at":"2026-07-05T00:45:47.918652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.02218","created_at":"2026-07-05T00:45:47.918652+00:00"},{"alias_kind":"pith_short_12","alias_value":"XPHPVTNMRTQU","created_at":"2026-07-05T00:45:47.918652+00:00"},{"alias_kind":"pith_short_16","alias_value":"XPHPVTNMRTQUB7QI","created_at":"2026-07-05T00:45:47.918652+00:00"},{"alias_kind":"pith_short_8","alias_value":"XPHPVTNM","created_at":"2026-07-05T00:45:47.918652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23364","citing_title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22053","citing_title":"Gradient-Descent Steps to Success over Mean Accuracy: A Paradigm Shift for ML","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18080","citing_title":"Edge Flow: A Tractable and Predictive Continuous-Time Model for Gradient Descent at the Edge of Stability","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04212","citing_title":"Edge of Stability Selectively Shapes Learning Across the Data Distribution","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00293","citing_title":"Accurate Large-sample Uncertainty Quantification using Stochastic Gradient Markov Chain Monte Carlo","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05219","citing_title":"Gradient Descent with Large Step Size Restores Symmetry in Deep Linear Networks with Multi-Pathway","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2501.02378","citing_title":"A ghost mechanism: An analytical model of abrupt learning in recurrent networks","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21292","citing_title":"Large-Step Training Dynamics of a Two-Factor Linear Transformer Model","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20005","citing_title":"Fine-Tuning Without Forgetting via Loss-Adaptive Learning Rates","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2102.01293","citing_title":"Scaling Laws for Transfer","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10079","citing_title":"Large Spikes in Stochastic Gradient Descent: A Large-Deviations View","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14200","citing_title":"How to Scale Mixture-of-Experts: From muP to the Maximally Scale-Stable Parameterization","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2010.14701","citing_title":"Scaling Laws for Autoregressive Generative Modeling","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10468","citing_title":"Can Muon Fine-tune Adam-Pretrained Models?","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21691","citing_title":"There Will Be a Scientific Theory of Deep Learning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00861","citing_title":"A General Language Assistant as a Laboratory for Alignment","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20446","citing_title":"The Origin of Edge of Stability","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04054","citing_title":"Endogenous Regime Switching Driven by Scalar-Irreducible Learning Dynamics","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06821","citing_title":"A Rod Flow Model for Adam at the Edge of Stability","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2207.05221","citing_title":"Language Models (Mostly) Know What They Know","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02968","citing_title":"Finite-Size Gradient Transport in Large Language Model Pretraining: From Cascade Size to Intensive Transport Efficiency","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13627","citing_title":"(How) Learning Rates Regulate Catastrophic Overtraining","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14669","citing_title":"Zeroth-Order Optimization at the Edge of Stability","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21016","citing_title":"SGD at the Edge of Stability: The Stochastic Sharpness Gap","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G","json":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G.json","graph_json":"https://pith.science/api/pith-number/XPHPVTNMRTQUB7QIDK4N53MI4G/graph.json","events_json":"https://pith.science/api/pith-number/XPHPVTNMRTQUB7QIDK4N53MI4G/events.json","paper":"https://pith.science/paper/XPHPVTNM"},"agent_actions":{"view_html":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G","download_json":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G.json","view_paper":"https://pith.science/paper/XPHPVTNM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.02218&json=true","fetch_graph":"https://pith.science/api/pith-number/XPHPVTNMRTQUB7QIDK4N53MI4G/graph.json","fetch_events":"https://pith.science/api/pith-number/XPHPVTNMRTQUB7QIDK4N53MI4G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G/action/storage_attestation","attest_author":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G/action/author_attestation","sign_citation":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G/action/citation_signature","submit_replication":"https://pith.science/pith/XPHPVTNMRTQUB7QIDK4N53MI4G/action/replication_record"}},"created_at":"2026-07-05T00:45:47.918652+00:00","updated_at":"2026-07-05T00:45:47.918652+00:00"}