{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IVYZGKYWW2PO22UQ5UHO57GQEX","short_pith_number":"pith:IVYZGKYW","schema_version":"1.0","canonical_sha256":"4571932b16b69eed6a90ed0eeefcd025c915962ad37cc62022aea118444af93a","source":{"kind":"arxiv","id":"2504.16041","version":1},"attestation_state":"computed","paper":{"title":"Muon Optimizer Accelerates Grokking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amund Tveit, Arve Skogvold, Bj{\\o}rn Remseth","submitted_at":"2025-04-22T17:08:09Z","abstract_excerpt":"This paper investigates the impact of different optimizers on the grokking phenomenon, where models exhibit delayed generalization. We conducted experiments across seven numerical tasks (primarily modular arithmetic) using a modern Transformer architecture. The experimental configuration systematically varied the optimizer (Muon vs. AdamW) and the softmax activation function (standard softmax, stablemax, and sparsemax) to assess their combined effect on learning dynamics. Our empirical evaluation reveals that the Muon optimizer, characterized by its use of spectral norm constraints and second-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16041","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-22T17:08:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ae322101a8019ffef09426a74d92e430abe7ed04bdc0fc459dbec8915b73b729","abstract_canon_sha256":"f83014f1f2f7e5aa47b4b529587f67b05774c4f4cf8539c4d50afb3fa4786cae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:34.333722Z","signature_b64":"fzwXDM6Pr/EJ3JmZ5RPZtz36VWmhvAazuGRhkIvAdgeea5ZIT+vrpkkHhPKWTqkHZnHaw49XuUxiHxQ7wuf/Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4571932b16b69eed6a90ed0eeefcd025c915962ad37cc62022aea118444af93a","last_reissued_at":"2026-07-05T10:52:34.333185Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:34.333185Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Muon Optimizer Accelerates Grokking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amund Tveit, Arve Skogvold, Bj{\\o}rn Remseth","submitted_at":"2025-04-22T17:08:09Z","abstract_excerpt":"This paper investigates the impact of different optimizers on the grokking phenomenon, where models exhibit delayed generalization. We conducted experiments across seven numerical tasks (primarily modular arithmetic) using a modern Transformer architecture. The experimental configuration systematically varied the optimizer (Muon vs. AdamW) and the softmax activation function (standard softmax, stablemax, and sparsemax) to assess their combined effect on learning dynamics. Our empirical evaluation reveals that the Muon optimizer, characterized by its use of spectral norm constraints and second-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16041","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16041","created_at":"2026-07-05T10:52:34.333247+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16041v1","created_at":"2026-07-05T10:52:34.333247+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16041","created_at":"2026-07-05T10:52:34.333247+00:00"},{"alias_kind":"pith_short_12","alias_value":"IVYZGKYWW2PO","created_at":"2026-07-05T10:52:34.333247+00:00"},{"alias_kind":"pith_short_16","alias_value":"IVYZGKYWW2PO22UQ","created_at":"2026-07-05T10:52:34.333247+00:00"},{"alias_kind":"pith_short_8","alias_value":"IVYZGKYW","created_at":"2026-07-05T10:52:34.333247+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26977","citing_title":"Convergence of Spectral Descent for Non-smooth Optimization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13079","citing_title":"Spectral Flattening Is All Muon Needs: How Orthogonalization Controls Learning Rate and Convergence","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06615","citing_title":"When and Why SignSGD Outperforms SGD: A Theoretical Study Based on $\\ell_1$-norm Lower Bounds","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX","json":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX.json","graph_json":"https://pith.science/api/pith-number/IVYZGKYWW2PO22UQ5UHO57GQEX/graph.json","events_json":"https://pith.science/api/pith-number/IVYZGKYWW2PO22UQ5UHO57GQEX/events.json","paper":"https://pith.science/paper/IVYZGKYW"},"agent_actions":{"view_html":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX","download_json":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX.json","view_paper":"https://pith.science/paper/IVYZGKYW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16041&json=true","fetch_graph":"https://pith.science/api/pith-number/IVYZGKYWW2PO22UQ5UHO57GQEX/graph.json","fetch_events":"https://pith.science/api/pith-number/IVYZGKYWW2PO22UQ5UHO57GQEX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX/action/storage_attestation","attest_author":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX/action/author_attestation","sign_citation":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX/action/citation_signature","submit_replication":"https://pith.science/pith/IVYZGKYWW2PO22UQ5UHO57GQEX/action/replication_record"}},"created_at":"2026-07-05T10:52:34.333247+00:00","updated_at":"2026-07-05T10:52:34.333247+00:00"}