{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:H2GPCGUEODXAUVKJAROSZFURR7","short_pith_number":"pith:H2GPCGUE","schema_version":"1.0","canonical_sha256":"3e8cf11a8470ee0a5549045d2c96918fca416043d0dddc3c751b84e241158413","source":{"kind":"arxiv","id":"2211.05729","version":2},"attestation_state":"computed","paper":{"title":"How Does Sharpness-Aware Minimization Minimize Sharpness?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kaiyue Wen, Tengyu Ma, Zhiyuan Li","submitted_at":"2022-11-10T17:56:38Z","abstract_excerpt":"Sharpness-Aware Minimization (SAM) is a highly effective regularization technique for improving the generalization of deep neural networks for various settings. However, the underlying working of SAM remains elusive because of various intriguing approximations in the theoretical characterizations. SAM intends to penalize a notion of sharpness of the model but implements a computationally efficient variant; moreover, a third notion of sharpness was used for proving generalization guarantees. The subtle differences in these notions of sharpness can indeed lead to significantly different empirica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.05729","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-10T17:56:38Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"b0e90bc82659d8ce690c3019dbb5aa979a3e8e8bef21a8e7913882fd2678122d","abstract_canon_sha256":"8f2ca555b0c0b6f78646d8d7b5b69347c95a81d920cfc5d2b112cb353e98dba1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:30:44.669400Z","signature_b64":"LgoVJgv+OE0U1ZgoebmSU+sOSLsg7FbhPMAble6U4sFkCS8aidaN3wAhbqyysP8Crrf9Ehj94IMWmuzQ5kq+Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3e8cf11a8470ee0a5549045d2c96918fca416043d0dddc3c751b84e241158413","last_reissued_at":"2026-07-05T05:30:44.668951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:30:44.668951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Does Sharpness-Aware Minimization Minimize Sharpness?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Kaiyue Wen, Tengyu Ma, Zhiyuan Li","submitted_at":"2022-11-10T17:56:38Z","abstract_excerpt":"Sharpness-Aware Minimization (SAM) is a highly effective regularization technique for improving the generalization of deep neural networks for various settings. However, the underlying working of SAM remains elusive because of various intriguing approximations in the theoretical characterizations. SAM intends to penalize a notion of sharpness of the model but implements a computationally efficient variant; moreover, a third notion of sharpness was used for proving generalization guarantees. The subtle differences in these notions of sharpness can indeed lead to significantly different empirica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.05729","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.05729/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.05729","created_at":"2026-07-05T05:30:44.669007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.05729v2","created_at":"2026-07-05T05:30:44.669007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.05729","created_at":"2026-07-05T05:30:44.669007+00:00"},{"alias_kind":"pith_short_12","alias_value":"H2GPCGUEODXA","created_at":"2026-07-05T05:30:44.669007+00:00"},{"alias_kind":"pith_short_16","alias_value":"H2GPCGUEODXAUVKJ","created_at":"2026-07-05T05:30:44.669007+00:00"},{"alias_kind":"pith_short_8","alias_value":"H2GPCGUE","created_at":"2026-07-05T05:30:44.669007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.21691","citing_title":"There Will Be a Scientific Theory of Deep Learning","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09258","citing_title":"Nexus: Same Pretraining Loss, Better Downstream Generalization via Common Minima","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07914","citing_title":"Flatness and Gradient Alignment Are Both Necessary: Spectral-Aware Gradient-Aligned Exploration for Multi-Distribution Learning","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7","json":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7.json","graph_json":"https://pith.science/api/pith-number/H2GPCGUEODXAUVKJAROSZFURR7/graph.json","events_json":"https://pith.science/api/pith-number/H2GPCGUEODXAUVKJAROSZFURR7/events.json","paper":"https://pith.science/paper/H2GPCGUE"},"agent_actions":{"view_html":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7","download_json":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7.json","view_paper":"https://pith.science/paper/H2GPCGUE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.05729&json=true","fetch_graph":"https://pith.science/api/pith-number/H2GPCGUEODXAUVKJAROSZFURR7/graph.json","fetch_events":"https://pith.science/api/pith-number/H2GPCGUEODXAUVKJAROSZFURR7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7/action/storage_attestation","attest_author":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7/action/author_attestation","sign_citation":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7/action/citation_signature","submit_replication":"https://pith.science/pith/H2GPCGUEODXAUVKJAROSZFURR7/action/replication_record"}},"created_at":"2026-07-05T05:30:44.669007+00:00","updated_at":"2026-07-05T05:30:44.669007+00:00"}