{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E67FCAEP55SLNG527TE7FH3ZZR","short_pith_number":"pith:E67FCAEP","schema_version":"1.0","canonical_sha256":"27be51008fef64b69bbafcc9f29f79cc66c3836b218c85a5b7befdaf07ef258f","source":{"kind":"arxiv","id":"2502.02407","version":1},"attestation_state":"computed","paper":{"title":"Avoiding spurious sharpness minimization broadens applicability of SAM","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Atish Agarwala, Hossein Mobahi, Sidak Pal Singh, Yann Dauphin","submitted_at":"2025-02-04T15:25:47Z","abstract_excerpt":"Curvature regularization techniques like Sharpness Aware Minimization (SAM) have shown great promise in improving generalization on vision tasks. However, we find that SAM performs poorly in domains like natural language processing (NLP), often degrading performance -- even with twice the compute budget. We investigate the discrepancy across domains and find that in the NLP setting, SAM is dominated by regularization of the logit statistics -- instead of improving the geometry of the function itself. We use this observation to develop an alternative algorithm we call Functional-SAM, which regu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02407","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-04T15:25:47Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"6bbf972afe1022ee0a51790c1c68b8eed5ea25c25184a829b5eac2acc1988cca","abstract_canon_sha256":"0b038ac9cd5a8f3dad38bda3a0134b18b00abccf612f9ba791bf1831ec00f059"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:28.358188Z","signature_b64":"wHGGeP3IHeXHgH3G2D7rMEhF3HrRU8nlrBvyVip7AX4BMBIaVlBs7mvFuuh8QXy7wdZilKtByMU8taqFLlzECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27be51008fef64b69bbafcc9f29f79cc66c3836b218c85a5b7befdaf07ef258f","last_reissued_at":"2026-07-05T10:09:28.357746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:28.357746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Avoiding spurious sharpness minimization broadens applicability of SAM","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Atish Agarwala, Hossein Mobahi, Sidak Pal Singh, Yann Dauphin","submitted_at":"2025-02-04T15:25:47Z","abstract_excerpt":"Curvature regularization techniques like Sharpness Aware Minimization (SAM) have shown great promise in improving generalization on vision tasks. However, we find that SAM performs poorly in domains like natural language processing (NLP), often degrading performance -- even with twice the compute budget. We investigate the discrepancy across domains and find that in the NLP setting, SAM is dominated by regularization of the logit statistics -- instead of improving the geometry of the function itself. We use this observation to develop an alternative algorithm we call Functional-SAM, which regu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02407","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02407","created_at":"2026-07-05T10:09:28.357805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02407v1","created_at":"2026-07-05T10:09:28.357805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02407","created_at":"2026-07-05T10:09:28.357805+00:00"},{"alias_kind":"pith_short_12","alias_value":"E67FCAEP55SL","created_at":"2026-07-05T10:09:28.357805+00:00"},{"alias_kind":"pith_short_16","alias_value":"E67FCAEP55SLNG52","created_at":"2026-07-05T10:09:28.357805+00:00"},{"alias_kind":"pith_short_8","alias_value":"E67FCAEP","created_at":"2026-07-05T10:09:28.357805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01827","citing_title":"Adaptive Sharpness-Aware Minimization with a Polyak-type Step size: A Theory-Grounded Scheduler","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR","json":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR.json","graph_json":"https://pith.science/api/pith-number/E67FCAEP55SLNG527TE7FH3ZZR/graph.json","events_json":"https://pith.science/api/pith-number/E67FCAEP55SLNG527TE7FH3ZZR/events.json","paper":"https://pith.science/paper/E67FCAEP"},"agent_actions":{"view_html":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR","download_json":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR.json","view_paper":"https://pith.science/paper/E67FCAEP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02407&json=true","fetch_graph":"https://pith.science/api/pith-number/E67FCAEP55SLNG527TE7FH3ZZR/graph.json","fetch_events":"https://pith.science/api/pith-number/E67FCAEP55SLNG527TE7FH3ZZR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR/action/storage_attestation","attest_author":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR/action/author_attestation","sign_citation":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR/action/citation_signature","submit_replication":"https://pith.science/pith/E67FCAEP55SLNG527TE7FH3ZZR/action/replication_record"}},"created_at":"2026-07-05T10:09:28.357805+00:00","updated_at":"2026-07-05T10:09:28.357805+00:00"}