{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PEBU5QBI52XSV6DWBIGVPVWLLJ","short_pith_number":"pith:PEBU5QBI","schema_version":"1.0","canonical_sha256":"79034ec028eeaf2af8760a0d57d6cb5a43b324c0fb52159fb1db08597d9a53ad","source":{"kind":"arxiv","id":"2210.04860","version":1},"attestation_state":"computed","paper":{"title":"Second-order regression models exhibit progressive sharpening to the edge of stability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Atish Agarwala, Fabian Pedregosa, Jeffrey Pennington","submitted_at":"2022-10-10T17:21:20Z","abstract_excerpt":"Recent studies of gradient descent with large step sizes have shown that there is often a regime with an initial increase in the largest eigenvalue of the loss Hessian (progressive sharpening), followed by a stabilization of the eigenvalue near the maximum value which allows convergence (edge of stability). These phenomena are intrinsically non-linear and do not happen for models in the constant Neural Tangent Kernel (NTK) regime, for which the predictive function is approximately linear in the parameters. As such, we consider the next simplest class of predictive models, namely those that are"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.04860","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-10T17:21:20Z","cross_cats_sorted":["cs.AI","math.OC"],"title_canon_sha256":"3e589acd63f427ea57d756f79e311e91aab5b10d00d2b065cb4413431581bacf","abstract_canon_sha256":"1160cab1647281295d4d0272e83d65696654f6dedab1d97ba4e206de3f8b2b09"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:57.362564Z","signature_b64":"5F2KqkSWUsqfWmOJEZmhze6rzxXs6iGJz+ZhbW1nXLk38bIxVQmAlgGmHOJYQ+xIzhxyho2lt96fHu/YsjAUCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79034ec028eeaf2af8760a0d57d6cb5a43b324c0fb52159fb1db08597d9a53ad","last_reissued_at":"2026-07-05T05:04:57.362132Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:57.362132Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Second-order regression models exhibit progressive sharpening to the edge of stability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Atish Agarwala, Fabian Pedregosa, Jeffrey Pennington","submitted_at":"2022-10-10T17:21:20Z","abstract_excerpt":"Recent studies of gradient descent with large step sizes have shown that there is often a regime with an initial increase in the largest eigenvalue of the loss Hessian (progressive sharpening), followed by a stabilization of the eigenvalue near the maximum value which allows convergence (edge of stability). These phenomena are intrinsically non-linear and do not happen for models in the constant Neural Tangent Kernel (NTK) regime, for which the predictive function is approximately linear in the parameters. As such, we consider the next simplest class of predictive models, namely those that are"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.04860","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.04860/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.04860","created_at":"2026-07-05T05:04:57.362200+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.04860v1","created_at":"2026-07-05T05:04:57.362200+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.04860","created_at":"2026-07-05T05:04:57.362200+00:00"},{"alias_kind":"pith_short_12","alias_value":"PEBU5QBI52XS","created_at":"2026-07-05T05:04:57.362200+00:00"},{"alias_kind":"pith_short_16","alias_value":"PEBU5QBI52XSV6DW","created_at":"2026-07-05T05:04:57.362200+00:00"},{"alias_kind":"pith_short_8","alias_value":"PEBU5QBI","created_at":"2026-07-05T05:04:57.362200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.20057","citing_title":"What Can Grokking Teach Us About Learning Under Nonstationarity?","ref_index":2017,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ","json":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ.json","graph_json":"https://pith.science/api/pith-number/PEBU5QBI52XSV6DWBIGVPVWLLJ/graph.json","events_json":"https://pith.science/api/pith-number/PEBU5QBI52XSV6DWBIGVPVWLLJ/events.json","paper":"https://pith.science/paper/PEBU5QBI"},"agent_actions":{"view_html":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ","download_json":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ.json","view_paper":"https://pith.science/paper/PEBU5QBI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.04860&json=true","fetch_graph":"https://pith.science/api/pith-number/PEBU5QBI52XSV6DWBIGVPVWLLJ/graph.json","fetch_events":"https://pith.science/api/pith-number/PEBU5QBI52XSV6DWBIGVPVWLLJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ/action/storage_attestation","attest_author":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ/action/author_attestation","sign_citation":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ/action/citation_signature","submit_replication":"https://pith.science/pith/PEBU5QBI52XSV6DWBIGVPVWLLJ/action/replication_record"}},"created_at":"2026-07-05T05:04:57.362200+00:00","updated_at":"2026-07-05T05:04:57.362200+00:00"}