{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7COUNJA2VF2C4ARW4ERNUR3YBU","short_pith_number":"pith:7COUNJA2","schema_version":"1.0","canonical_sha256":"f89d46a41aa9742e0236e122da47780d2ac3905ac77facad8e45cc2445e7d461","source":{"kind":"arxiv","id":"2502.02900","version":2},"attestation_state":"computed","paper":{"title":"A Note on the Convergence of Muon","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Jiaxiang Li, Mingyi Hong","submitted_at":"2025-02-05T05:44:22Z","abstract_excerpt":"In this note, we inspect the convergence of a new optimizer for pretraining LLMs, namely the Muon optimizer. Such an optimizer is closely related to a specialized steepest descent method where the update direction is the minimizer of the quadratic approximation of the objective function under spectral norm. We provide the convergence analysis on both versions of the optimizer and discuss its implications."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02900","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2025-02-05T05:44:22Z","cross_cats_sorted":[],"title_canon_sha256":"aadbdf3859aadeee1f5c5ef3c8aee283e5877cc09b2861838f91cea02d462d21","abstract_canon_sha256":"505b473b018c923106d182d0bc8c316231d52585b1a4f393f5b7029f74e9035d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:19.815072Z","signature_b64":"elqLSLqQPulBBQ0Vbr86G9hj89KcCmG88BvUNi/AX2FyZm49nyU4y0CkALnj6F/VGp6zmqad7I5J8BPlTjwEBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f89d46a41aa9742e0236e122da47780d2ac3905ac77facad8e45cc2445e7d461","last_reissued_at":"2026-07-05T11:13:19.814600Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:19.814600Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Note on the Convergence of Muon","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"math.OC","authors_text":"Jiaxiang Li, Mingyi Hong","submitted_at":"2025-02-05T05:44:22Z","abstract_excerpt":"In this note, we inspect the convergence of a new optimizer for pretraining LLMs, namely the Muon optimizer. Such an optimizer is closely related to a specialized steepest descent method where the update direction is the minimizer of the quadratic approximation of the objective function under spectral norm. We provide the convergence analysis on both versions of the optimizer and discuss its implications."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02900","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02900/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02900","created_at":"2026-07-05T11:13:19.814661+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02900v2","created_at":"2026-07-05T11:13:19.814661+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02900","created_at":"2026-07-05T11:13:19.814661+00:00"},{"alias_kind":"pith_short_12","alias_value":"7COUNJA2VF2C","created_at":"2026-07-05T11:13:19.814661+00:00"},{"alias_kind":"pith_short_16","alias_value":"7COUNJA2VF2C4ARW","created_at":"2026-07-05T11:13:19.814661+00:00"},{"alias_kind":"pith_short_8","alias_value":"7COUNJA2","created_at":"2026-07-05T11:13:19.814661+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18528","citing_title":"Scale-Invariant Neural Network Optimization: Norm Geometry and Heavy-Tailed Noise","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19811","citing_title":"LionMuon: Alternating Spectral and Sign Descent for Efficient Training","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26977","citing_title":"Convergence of Spectral Descent for Non-smooth Optimization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23061","citing_title":"Anytime Training with Schedule-Free Spectral Optimization","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19619","citing_title":"MiMuon: Mixed Muon Optimizer with Improved Generalization for Large Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19811","citing_title":"LionMuon: Alternating Spectral and Sign Descent for Efficient Training","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23737","citing_title":"On the Convergence Analysis of Muon","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13079","citing_title":"Spectral Flattening Is All Muon Needs: How Orthogonalization Controls Learning Rate and Convergence","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08980","citing_title":"Muon Does Not Converge on Convex Lipschitz Functions","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09238","citing_title":"Intrinsic Muon: Spectral Optimization on Riemannian Matrix Manifolds","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06615","citing_title":"When and Why SignSGD Outperforms SGD: A Theoretical Study Based on $\\ell_1$-norm Lower Bounds","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21616","citing_title":"Convergence Rate Analysis of SOAP with Arbitrary Orthogonal Projection Matrices","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06884","citing_title":"Muon with Nesterov Momentum: Heavy-Tailed Noise and (Randomized) Inexact Polar Decomposition","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04726","citing_title":"A Muon-Accelerated Algorithm for Low Separation Rank Tensor Generalized Linear Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17423","citing_title":"A unified convergence theory for adaptive first-order methods in the nonconvex case, including AdaNorm, full and diagonal AdaGrad, Shampoo and Muo","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU","json":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU.json","graph_json":"https://pith.science/api/pith-number/7COUNJA2VF2C4ARW4ERNUR3YBU/graph.json","events_json":"https://pith.science/api/pith-number/7COUNJA2VF2C4ARW4ERNUR3YBU/events.json","paper":"https://pith.science/paper/7COUNJA2"},"agent_actions":{"view_html":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU","download_json":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU.json","view_paper":"https://pith.science/paper/7COUNJA2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02900&json=true","fetch_graph":"https://pith.science/api/pith-number/7COUNJA2VF2C4ARW4ERNUR3YBU/graph.json","fetch_events":"https://pith.science/api/pith-number/7COUNJA2VF2C4ARW4ERNUR3YBU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU/action/storage_attestation","attest_author":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU/action/author_attestation","sign_citation":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU/action/citation_signature","submit_replication":"https://pith.science/pith/7COUNJA2VF2C4ARW4ERNUR3YBU/action/replication_record"}},"created_at":"2026-07-05T11:13:19.814661+00:00","updated_at":"2026-07-05T11:13:19.814661+00:00"}