{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6KV3STARSKM2X2SLZFDLVAVBFA","short_pith_number":"pith:6KV3STAR","schema_version":"1.0","canonical_sha256":"f2abb94c119299abea4bc946ba82a12808c85e8f985ae0e49fe59c815b3f6d14","source":{"kind":"arxiv","id":"2509.02981","version":2},"attestation_state":"computed","paper":{"title":"AdaGrad Meets Muon: Adaptive Stepsizes for Orthogonal Updates","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Hayden Schaeffer, Minxin Zhang, Yuxuan Liu","submitted_at":"2025-09-03T03:42:22Z","abstract_excerpt":"The recently proposed Muon optimizer updates weight matrices via orthogonalized momentum and has demonstrated strong empirical success in large language model training. However, it remains unclear how to determine the learning rates for such orthogonalized updates. AdaGrad, by contrast, is a widely used adaptive method that scales stochastic gradients by accumulated past gradients. We propose a new algorithm, AdaGO, which combines a norm-based AdaGrad-type stepsize with an orthogonalized update direction, bringing together the benefits of both approaches. Unlike other adaptive variants of Muon"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.02981","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-09-03T03:42:22Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"019abf23ef097615c832d773cb54563e213029b0d97d202ad9c6199f2174ead5","abstract_canon_sha256":"302c147a54872ff42186bba3d3947721f26135a3ed09ae5c80e17bb901a94d04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:06:20.026793Z","signature_b64":"+Qf9ILr9IJeR2YdzvH0SYFV6OHc8lgqiNx+gxsG4T7P1ddPs+bJSr8u+PqayNVIpcUYri+uEGWaglwQyhvJhCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2abb94c119299abea4bc946ba82a12808c85e8f985ae0e49fe59c815b3f6d14","last_reissued_at":"2026-07-05T12:06:20.026263Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:06:20.026263Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaGrad Meets Muon: Adaptive Stepsizes for Orthogonal Updates","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Hayden Schaeffer, Minxin Zhang, Yuxuan Liu","submitted_at":"2025-09-03T03:42:22Z","abstract_excerpt":"The recently proposed Muon optimizer updates weight matrices via orthogonalized momentum and has demonstrated strong empirical success in large language model training. However, it remains unclear how to determine the learning rates for such orthogonalized updates. AdaGrad, by contrast, is a widely used adaptive method that scales stochastic gradients by accumulated past gradients. We propose a new algorithm, AdaGO, which combines a norm-based AdaGrad-type stepsize with an orthogonalized update direction, bringing together the benefits of both approaches. Unlike other adaptive variants of Muon"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.02981","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.02981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.02981","created_at":"2026-07-05T12:06:20.026329+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.02981v2","created_at":"2026-07-05T12:06:20.026329+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.02981","created_at":"2026-07-05T12:06:20.026329+00:00"},{"alias_kind":"pith_short_12","alias_value":"6KV3STARSKM2","created_at":"2026-07-05T12:06:20.026329+00:00"},{"alias_kind":"pith_short_16","alias_value":"6KV3STARSKM2X2SL","created_at":"2026-07-05T12:06:20.026329+00:00"},{"alias_kind":"pith_short_8","alias_value":"6KV3STAR","created_at":"2026-07-05T12:06:20.026329+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08783","citing_title":"OptMuon: Closed-Loop Orthogonalized Momentum Methods for Stochastic Optimization with Zero-Noise Optimality","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01787","citing_title":"Stochastic convergence of parallel asynchronous adaptive first-order methods","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":173,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26977","citing_title":"Convergence of Spectral Descent for Non-smooth Optimization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08783","citing_title":"OptMuon: Closed-Loop Orthogonalized Momentum Methods for Stochastic Optimization with Zero-Noise Optimality","ref_index":195,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18106","citing_title":"Symmetry-Compatible Principle for Optimizer Design: Embeddings, LM Heads, SwiGLU MLPs, and MoE Routers","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11850","citing_title":"Constrained Stochastic Spectral Preconditioning Converges for Nonconvex Objectives","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06615","citing_title":"When and Why SignSGD Outperforms SGD: A Theoretical Study Based on $\\ell_1$-norm Lower Bounds","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04726","citing_title":"A Muon-Accelerated Algorithm for Low Separation Rank Tensor Generalized Linear Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17423","citing_title":"A unified convergence theory for adaptive first-order methods in the nonconvex case, including AdaNorm, full and diagonal AdaGrad, Shampoo and Muo","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA","json":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA.json","graph_json":"https://pith.science/api/pith-number/6KV3STARSKM2X2SLZFDLVAVBFA/graph.json","events_json":"https://pith.science/api/pith-number/6KV3STARSKM2X2SLZFDLVAVBFA/events.json","paper":"https://pith.science/paper/6KV3STAR"},"agent_actions":{"view_html":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA","download_json":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA.json","view_paper":"https://pith.science/paper/6KV3STAR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.02981&json=true","fetch_graph":"https://pith.science/api/pith-number/6KV3STARSKM2X2SLZFDLVAVBFA/graph.json","fetch_events":"https://pith.science/api/pith-number/6KV3STARSKM2X2SLZFDLVAVBFA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA/action/storage_attestation","attest_author":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA/action/author_attestation","sign_citation":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA/action/citation_signature","submit_replication":"https://pith.science/pith/6KV3STARSKM2X2SLZFDLVAVBFA/action/replication_record"}},"created_at":"2026-07-05T12:06:20.026329+00:00","updated_at":"2026-07-05T12:06:20.026329+00:00"}