{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CQAHT5DKVZKDE22V3MBQFMGPK4","short_pith_number":"pith:CQAHT5DK","schema_version":"1.0","canonical_sha256":"140079f46aae54326b55db0302b0cf5703e5e4b7a68923af4addf2344fa614f5","source":{"kind":"arxiv","id":"2410.11289","version":2},"attestation_state":"computed","paper":{"title":"Subspace Optimization for Large Language Models with Convergence Guarantees","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Chuyan Chen, Kun Yuan, Pengrui Li, Yipeng Hu, Yutong He","submitted_at":"2024-10-15T05:16:32Z","abstract_excerpt":"Subspace optimization algorithms, such as GaLore (Zhao et al., 2024), have gained attention for pre-training and fine-tuning large language models (LLMs) due to their memory efficiency. However, their convergence guarantees remain unclear, particularly in stochastic settings. In this paper, we reveal that GaLore does not always converge to the optimal solution and provide an explicit counterexample to support this finding. We further explore the conditions under which GaLore achieves convergence, showing that it does so when either (i) a sufficiently large mini-batch size is used or (ii) the g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11289","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-15T05:16:32Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"c5aeea46e650b5aea9f949810523ce32432ed186e8adfe1509a34cc81dd469b6","abstract_canon_sha256":"44f093caf585a3ec2359babf23bf5eba8ca3ed8d2866216477c2ffed197f9bd1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:39.797869Z","signature_b64":"e7A1qqt9ocR+8G/C8jcOpVo70wn1YxgYqeJrFootSAGVb7vGSZj1JUtUy2OG40Ju57wTmrPn+IEp2tQjNUQ+Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"140079f46aae54326b55db0302b0cf5703e5e4b7a68923af4addf2344fa614f5","last_reissued_at":"2026-07-05T11:15:39.797326Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:39.797326Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Subspace Optimization for Large Language Models with Convergence Guarantees","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Chuyan Chen, Kun Yuan, Pengrui Li, Yipeng Hu, Yutong He","submitted_at":"2024-10-15T05:16:32Z","abstract_excerpt":"Subspace optimization algorithms, such as GaLore (Zhao et al., 2024), have gained attention for pre-training and fine-tuning large language models (LLMs) due to their memory efficiency. However, their convergence guarantees remain unclear, particularly in stochastic settings. In this paper, we reveal that GaLore does not always converge to the optimal solution and provide an explicit counterexample to support this finding. We further explore the conditions under which GaLore achieves convergence, showing that it does so when either (i) a sufficiently large mini-batch size is used or (ii) the g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11289","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11289","created_at":"2026-07-05T11:15:39.797411+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11289v2","created_at":"2026-07-05T11:15:39.797411+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11289","created_at":"2026-07-05T11:15:39.797411+00:00"},{"alias_kind":"pith_short_12","alias_value":"CQAHT5DKVZKD","created_at":"2026-07-05T11:15:39.797411+00:00"},{"alias_kind":"pith_short_16","alias_value":"CQAHT5DKVZKDE22V","created_at":"2026-07-05T11:15:39.797411+00:00"},{"alias_kind":"pith_short_8","alias_value":"CQAHT5DK","created_at":"2026-07-05T11:15:39.797411+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05872","citing_title":"No Subspace to Track: Non-Identifiability and Optimizer State in Low-Rank Training","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2501.07237","citing_title":"GWT: Scalable Optimizer State Compression for Large Language Model Training","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15588","citing_title":"Memory-Efficient Differentially Private Training with Gradient Random Projection","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18993","citing_title":"CR-Net: Scaling Parameter-Efficient Training with Cross-Layer Low-Rank Structure","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06316","citing_title":"Pro-KLShampoo: Projected KL-Shampoo with Whitening Recovered by Orthogonalization","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4","json":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4.json","graph_json":"https://pith.science/api/pith-number/CQAHT5DKVZKDE22V3MBQFMGPK4/graph.json","events_json":"https://pith.science/api/pith-number/CQAHT5DKVZKDE22V3MBQFMGPK4/events.json","paper":"https://pith.science/paper/CQAHT5DK"},"agent_actions":{"view_html":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4","download_json":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4.json","view_paper":"https://pith.science/paper/CQAHT5DK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11289&json=true","fetch_graph":"https://pith.science/api/pith-number/CQAHT5DKVZKDE22V3MBQFMGPK4/graph.json","fetch_events":"https://pith.science/api/pith-number/CQAHT5DKVZKDE22V3MBQFMGPK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4/action/storage_attestation","attest_author":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4/action/author_attestation","sign_citation":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4/action/citation_signature","submit_replication":"https://pith.science/pith/CQAHT5DKVZKDE22V3MBQFMGPK4/action/replication_record"}},"created_at":"2026-07-05T11:15:39.797411+00:00","updated_at":"2026-07-05T11:15:39.797411+00:00"}