{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D72PKYKK4HQIFO3ZEZAUAY763V","short_pith_number":"pith:D72PKYKK","schema_version":"1.0","canonical_sha256":"1ff4f5614ae1e082bb7926414063fedd55f780d828f0f152099d46ad5aa1f979","source":{"kind":"arxiv","id":"2411.00075","version":2},"attestation_state":"computed","paper":{"title":"{\\mu}P$^2$: Effective Sharpness Aware Minimization Requires Layerwise Perturbation Scaling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jin Xu, Leena Chennuru Vankadara, Moritz Haas, Volkan Cevher","submitted_at":"2024-10-31T16:32:04Z","abstract_excerpt":"Sharpness Aware Minimization (SAM) enhances performance across various neural architectures and datasets. As models are continually scaled up to improve performance, a rigorous understanding of SAM's scaling behaviour is paramount. To this end, we study the infinite-width limit of neural networks trained with SAM, using the Tensor Programs framework. Our findings reveal that the dynamics of standard SAM effectively reduce to applying SAM solely in the last layer in wide neural networks, even with optimal hyperparameters. In contrast, we identify a stable parameterization with layerwise perturb"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.00075","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-31T16:32:04Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"784d2819f42b66319f3d465408cf9deb8398c09b217e3497e8f996c659df894f","abstract_canon_sha256":"dffadc924ec65ec81c1ad485419d2839687a8f6432c1e50233aef23964a09793"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:22.825395Z","signature_b64":"VHgjPRipi/LesZyTuQCPx4Xquvps+Wi1zqvIUThlkJ32ZTb7oEySGEGUPPTNUIESKBHf2APSK+5VpJvW286qDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ff4f5614ae1e082bb7926414063fedd55f780d828f0f152099d46ad5aa1f979","last_reissued_at":"2026-07-05T10:12:22.824885Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:22.824885Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"{\\mu}P$^2$: Effective Sharpness Aware Minimization Requires Layerwise Perturbation Scaling","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jin Xu, Leena Chennuru Vankadara, Moritz Haas, Volkan Cevher","submitted_at":"2024-10-31T16:32:04Z","abstract_excerpt":"Sharpness Aware Minimization (SAM) enhances performance across various neural architectures and datasets. As models are continually scaled up to improve performance, a rigorous understanding of SAM's scaling behaviour is paramount. To this end, we study the infinite-width limit of neural networks trained with SAM, using the Tensor Programs framework. Our findings reveal that the dynamics of standard SAM effectively reduce to applying SAM solely in the last layer in wide neural networks, even with optimal hyperparameters. In contrast, we identify a stable parameterization with layerwise perturb"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.00075","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.00075/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.00075","created_at":"2026-07-05T10:12:22.824944+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.00075v2","created_at":"2026-07-05T10:12:22.824944+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.00075","created_at":"2026-07-05T10:12:22.824944+00:00"},{"alias_kind":"pith_short_12","alias_value":"D72PKYKK4HQI","created_at":"2026-07-05T10:12:22.824944+00:00"},{"alias_kind":"pith_short_16","alias_value":"D72PKYKK4HQIFO3Z","created_at":"2026-07-05T10:12:22.824944+00:00"},{"alias_kind":"pith_short_8","alias_value":"D72PKYKK","created_at":"2026-07-05T10:12:22.824944+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02407","citing_title":"Avoiding spurious sharpness minimization broadens applicability of SAM","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V","json":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V.json","graph_json":"https://pith.science/api/pith-number/D72PKYKK4HQIFO3ZEZAUAY763V/graph.json","events_json":"https://pith.science/api/pith-number/D72PKYKK4HQIFO3ZEZAUAY763V/events.json","paper":"https://pith.science/paper/D72PKYKK"},"agent_actions":{"view_html":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V","download_json":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V.json","view_paper":"https://pith.science/paper/D72PKYKK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.00075&json=true","fetch_graph":"https://pith.science/api/pith-number/D72PKYKK4HQIFO3ZEZAUAY763V/graph.json","fetch_events":"https://pith.science/api/pith-number/D72PKYKK4HQIFO3ZEZAUAY763V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V/action/storage_attestation","attest_author":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V/action/author_attestation","sign_citation":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V/action/citation_signature","submit_replication":"https://pith.science/pith/D72PKYKK4HQIFO3ZEZAUAY763V/action/replication_record"}},"created_at":"2026-07-05T10:12:22.824944+00:00","updated_at":"2026-07-05T10:12:22.824944+00:00"}