{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CURN2RUJE2DQZNSXH7MTJELMGL","short_pith_number":"pith:CURN2RUJ","schema_version":"1.0","canonical_sha256":"1522dd468926870cb6573fd934916c32e758acaf8a6ab4298d70fa90b4c91fd9","source":{"kind":"arxiv","id":"2207.14484","version":2},"attestation_state":"computed","paper":{"title":"Adaptive Gradient Methods at the Edge of Stability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Behrooz Ghorbani, Daniel Suo, David Cardoze, George E. Dahl, Jeremy M. Cohen, Justin Gilmer, Michal Badura, Naman Agarwal, Shankar Krishnan, Sourabh Medapati, Zachary Nado","submitted_at":"2022-07-29T05:23:47Z","abstract_excerpt":"Very little is known about the training dynamics of adaptive gradient methods like Adam in deep learning. In this paper, we shed light on the behavior of these algorithms in the full-batch and sufficiently large batch settings. Specifically, we empirically demonstrate that during full-batch training, the maximum eigenvalue of the preconditioned Hessian typically equilibrates at a certain numerical value -- the stability threshold of a gradient descent algorithm. For Adam with step size $\\eta$ and $\\beta_1 = 0.9$, this stability threshold is $38/\\eta$. Similar effects occur during minibatch tra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.14484","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-07-29T05:23:47Z","cross_cats_sorted":[],"title_canon_sha256":"26213289254e06676c331c2ad657437cdd156f570a151068fcfd9cbdfaf206dd","abstract_canon_sha256":"4d68b066b952acb43c31b75003940d4b11e6e7e1b06cac7a9b21a562d14733dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:15.200495Z","signature_b64":"RPPS27JN+Q/NUnxeK9vFuSAIvwz79KDPRhcQHutFjUb4tt5ZS4xl/fVSeZgzqRFF3ecuiHYjxQE7/K8jGUrOAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1522dd468926870cb6573fd934916c32e758acaf8a6ab4298d70fa90b4c91fd9","last_reissued_at":"2026-07-05T08:08:15.199979Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:15.199979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Gradient Methods at the Edge of Stability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Behrooz Ghorbani, Daniel Suo, David Cardoze, George E. Dahl, Jeremy M. Cohen, Justin Gilmer, Michal Badura, Naman Agarwal, Shankar Krishnan, Sourabh Medapati, Zachary Nado","submitted_at":"2022-07-29T05:23:47Z","abstract_excerpt":"Very little is known about the training dynamics of adaptive gradient methods like Adam in deep learning. In this paper, we shed light on the behavior of these algorithms in the full-batch and sufficiently large batch settings. Specifically, we empirically demonstrate that during full-batch training, the maximum eigenvalue of the preconditioned Hessian typically equilibrates at a certain numerical value -- the stability threshold of a gradient descent algorithm. For Adam with step size $\\eta$ and $\\beta_1 = 0.9$, this stability threshold is $38/\\eta$. Similar effects occur during minibatch tra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.14484","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.14484/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.14484","created_at":"2026-07-05T08:08:15.200040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.14484v2","created_at":"2026-07-05T08:08:15.200040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.14484","created_at":"2026-07-05T08:08:15.200040+00:00"},{"alias_kind":"pith_short_12","alias_value":"CURN2RUJE2DQ","created_at":"2026-07-05T08:08:15.200040+00:00"},{"alias_kind":"pith_short_16","alias_value":"CURN2RUJE2DQZNSX","created_at":"2026-07-05T08:08:15.200040+00:00"},{"alias_kind":"pith_short_8","alias_value":"CURN2RUJ","created_at":"2026-07-05T08:08:15.200040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19179","citing_title":"Compute Efficiency and Serial Runtime Tradeoffs for Stochastic Momentum Methods","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00510","citing_title":"Prototype Language Models","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2505.13196","citing_title":"A Physics-Inspired Optimizer: Velocity Regularized Adam","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07870","citing_title":"Spectral Dynamics in Deep Networks: Feature Learning, Outlier Escape, and Learning Rate Transfer","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2505.24275","citing_title":"GradPower: Powering Gradients for Faster Language Model Pre-Training","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16622","citing_title":"Does Weight Decay Enhance Training Stability?","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14200","citing_title":"How to Scale Mixture-of-Experts: From muP to the Maximally Scale-Stable Parameterization","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09552","citing_title":"Phases of Muon: When Muon Eclipses SignSGD","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06821","citing_title":"A Rod Flow Model for Adam at the Edge of Stability","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07870","citing_title":"Spectral Dynamics in Deep Networks: Feature Learning, Outlier Escape, and Learning Rate Transfer","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14108","citing_title":"Momentum Further Constrains Sharpness at the Edge of Stochastic Stability","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14669","citing_title":"Zeroth-Order Optimization at the Edge of Stability","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL","json":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL.json","graph_json":"https://pith.science/api/pith-number/CURN2RUJE2DQZNSXH7MTJELMGL/graph.json","events_json":"https://pith.science/api/pith-number/CURN2RUJE2DQZNSXH7MTJELMGL/events.json","paper":"https://pith.science/paper/CURN2RUJ"},"agent_actions":{"view_html":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL","download_json":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL.json","view_paper":"https://pith.science/paper/CURN2RUJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.14484&json=true","fetch_graph":"https://pith.science/api/pith-number/CURN2RUJE2DQZNSXH7MTJELMGL/graph.json","fetch_events":"https://pith.science/api/pith-number/CURN2RUJE2DQZNSXH7MTJELMGL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL/action/storage_attestation","attest_author":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL/action/author_attestation","sign_citation":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL/action/citation_signature","submit_replication":"https://pith.science/pith/CURN2RUJE2DQZNSXH7MTJELMGL/action/replication_record"}},"created_at":"2026-07-05T08:08:15.200040+00:00","updated_at":"2026-07-05T08:08:15.200040+00:00"}