{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZPKT2ZYAAISEBO2JUOJ3KUQSGJ","short_pith_number":"pith:ZPKT2ZYA","schema_version":"1.0","canonical_sha256":"cbd53d6700022440bb49a393b55212327ba67221f0bf050005bb47613d7a083a","source":{"kind":"arxiv","id":"2310.17087","version":2},"attestation_state":"computed","paper":{"title":"Good regularity creates large learning rate implicit biases: edge of stability, balancing, and catapult","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.DS","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Molei Tao, Tuo Zhao, Yuqing Wang, Zhenghao Xu","submitted_at":"2023-10-26T01:11:17Z","abstract_excerpt":"Large learning rates, when applied to gradient descent for nonconvex optimization, yield various implicit biases including the edge of stability (Cohen et al., 2021), balancing (Wang et al., 2022), and catapult (Lewkowycz et al., 2020). These phenomena cannot be well explained by classical optimization theory. Though significant theoretical progress has been made in understanding these implicit biases, it remains unclear for which objective functions would they be more likely. This paper provides an initial step in answering this question and also shows that these implicit biases are in fact v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17087","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-26T01:11:17Z","cross_cats_sorted":["math.DS","math.OC","stat.ML"],"title_canon_sha256":"2df60af4059079e2275083c1ce70dd52fa207375d82c9e37f6d6fdfa0f243ec3","abstract_canon_sha256":"5c6dbe9148db167553df6d23c4e16084280b2f467b9f31bc91e54c8f22a26faa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:23:09.129234Z","signature_b64":"UCni1PFJSmJrMTuRErawD9nQL2BSkrGJa48H8U+IBoMBQDNudvLznEACEts3f7dbkNAFtdmYcZfRzHB0xqI8Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cbd53d6700022440bb49a393b55212327ba67221f0bf050005bb47613d7a083a","last_reissued_at":"2026-07-05T07:23:09.128752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:23:09.128752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Good regularity creates large learning rate implicit biases: edge of stability, balancing, and catapult","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.DS","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Molei Tao, Tuo Zhao, Yuqing Wang, Zhenghao Xu","submitted_at":"2023-10-26T01:11:17Z","abstract_excerpt":"Large learning rates, when applied to gradient descent for nonconvex optimization, yield various implicit biases including the edge of stability (Cohen et al., 2021), balancing (Wang et al., 2022), and catapult (Lewkowycz et al., 2020). These phenomena cannot be well explained by classical optimization theory. Though significant theoretical progress has been made in understanding these implicit biases, it remains unclear for which objective functions would they be more likely. This paper provides an initial step in answering this question and also shows that these implicit biases are in fact v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17087","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17087/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17087","created_at":"2026-07-05T07:23:09.128810+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17087v2","created_at":"2026-07-05T07:23:09.128810+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17087","created_at":"2026-07-05T07:23:09.128810+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZPKT2ZYAAISE","created_at":"2026-07-05T07:23:09.128810+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZPKT2ZYAAISEBO2J","created_at":"2026-07-05T07:23:09.128810+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZPKT2ZYA","created_at":"2026-07-05T07:23:09.128810+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23364","citing_title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21016","citing_title":"SGD at the Edge of Stability: The Stochastic Sharpness Gap","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ","json":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ.json","graph_json":"https://pith.science/api/pith-number/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/graph.json","events_json":"https://pith.science/api/pith-number/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/events.json","paper":"https://pith.science/paper/ZPKT2ZYA"},"agent_actions":{"view_html":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ","download_json":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ.json","view_paper":"https://pith.science/paper/ZPKT2ZYA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17087&json=true","fetch_graph":"https://pith.science/api/pith-number/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/graph.json","fetch_events":"https://pith.science/api/pith-number/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/action/storage_attestation","attest_author":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/action/author_attestation","sign_citation":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/action/citation_signature","submit_replication":"https://pith.science/pith/ZPKT2ZYAAISEBO2JUOJ3KUQSGJ/action/replication_record"}},"created_at":"2026-07-05T07:23:09.128810+00:00","updated_at":"2026-07-05T07:23:09.128810+00:00"}