{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5N6DMRSOYSJRDSROGTFEVPZVXE","short_pith_number":"pith:5N6DMRSO","schema_version":"1.0","canonical_sha256":"eb7c36464ec49311ca2e34ca4abf35b900ec53687d9988478e4c99070adc937f","source":{"kind":"arxiv","id":"2305.11788","version":2},"attestation_state":"computed","paper":{"title":"Implicit Bias of Gradient Descent for Logistic Regression at the Edge of Stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jason D. Lee, Jingfeng Wu, Vladimir Braverman","submitted_at":"2023-05-19T16:24:47Z","abstract_excerpt":"Recent research has observed that in machine learning optimization, gradient descent (GD) often operates at the edge of stability (EoS) [Cohen, et al., 2021], where the stepsizes are set to be large, resulting in non-monotonic losses induced by the GD iterates. This paper studies the convergence and implicit bias of constant-stepsize GD for logistic regression on linearly separable data in the EoS regime. Despite the presence of local oscillations, we prove that the logistic loss can be minimized by GD with \\emph{any} constant stepsize over a long time scale. Furthermore, we prove that with \\e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.11788","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-19T16:24:47Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"bb01a640b5071ecbfa42370aa4b81041d5279dbd3fd10140b010d003261bd397","abstract_canon_sha256":"4158cf6b46565c19c18c3ac448a0a0bff31aaaf0fc3a8196b0c6dad8d6153741"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:01.356344Z","signature_b64":"FPreNU+6mngTZyqtUjHWjWOLUX3uZ05xnI391/UCg22PFTn6TVqaSVgUNprlu9dcBH8BzG3e+5+2vlnXZwM9Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb7c36464ec49311ca2e34ca4abf35b900ec53687d9988478e4c99070adc937f","last_reissued_at":"2026-07-05T07:01:01.355883Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:01.355883Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Implicit Bias of Gradient Descent for Logistic Regression at the Edge of Stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jason D. Lee, Jingfeng Wu, Vladimir Braverman","submitted_at":"2023-05-19T16:24:47Z","abstract_excerpt":"Recent research has observed that in machine learning optimization, gradient descent (GD) often operates at the edge of stability (EoS) [Cohen, et al., 2021], where the stepsizes are set to be large, resulting in non-monotonic losses induced by the GD iterates. This paper studies the convergence and implicit bias of constant-stepsize GD for logistic regression on linearly separable data in the EoS regime. Despite the presence of local oscillations, we prove that the logistic loss can be minimized by GD with \\emph{any} constant stepsize over a long time scale. Furthermore, we prove that with \\e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.11788","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.11788/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.11788","created_at":"2026-07-05T07:01:01.355934+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.11788v2","created_at":"2026-07-05T07:01:01.355934+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.11788","created_at":"2026-07-05T07:01:01.355934+00:00"},{"alias_kind":"pith_short_12","alias_value":"5N6DMRSOYSJR","created_at":"2026-07-05T07:01:01.355934+00:00"},{"alias_kind":"pith_short_16","alias_value":"5N6DMRSOYSJRDSRO","created_at":"2026-07-05T07:01:01.355934+00:00"},{"alias_kind":"pith_short_8","alias_value":"5N6DMRSO","created_at":"2026-07-05T07:01:01.355934+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.11228","citing_title":"Gradient Descent on Logistic Regression: Do Large Step-Sizes Work with Data on the Sphere?","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE","json":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE.json","graph_json":"https://pith.science/api/pith-number/5N6DMRSOYSJRDSROGTFEVPZVXE/graph.json","events_json":"https://pith.science/api/pith-number/5N6DMRSOYSJRDSROGTFEVPZVXE/events.json","paper":"https://pith.science/paper/5N6DMRSO"},"agent_actions":{"view_html":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE","download_json":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE.json","view_paper":"https://pith.science/paper/5N6DMRSO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.11788&json=true","fetch_graph":"https://pith.science/api/pith-number/5N6DMRSOYSJRDSROGTFEVPZVXE/graph.json","fetch_events":"https://pith.science/api/pith-number/5N6DMRSOYSJRDSROGTFEVPZVXE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE/action/storage_attestation","attest_author":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE/action/author_attestation","sign_citation":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE/action/citation_signature","submit_replication":"https://pith.science/pith/5N6DMRSOYSJRDSROGTFEVPZVXE/action/replication_record"}},"created_at":"2026-07-05T07:01:01.355934+00:00","updated_at":"2026-07-05T07:01:01.355934+00:00"}