{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6NZK6ZFVGK2D34RX37DVGS6D3Q","short_pith_number":"pith:6NZK6ZFV","schema_version":"1.0","canonical_sha256":"f372af64b532b43df237dfc7534bc3dc0586a97234563d7291ef080d47cec2b6","source":{"kind":"arxiv","id":"2502.07222","version":1},"attestation_state":"computed","paper":{"title":"A Memory Efficient Randomized Subspace Optimization Method for Training Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kun Yuan, Yiming Chen, Yin Liu, Yuan Zhang, Zaiwen Wen","submitted_at":"2025-02-11T03:32:10Z","abstract_excerpt":"The memory challenges associated with training Large Language Models (LLMs) have become a critical concern, particularly when using the Adam optimizer. To address this issue, numerous memory-efficient techniques have been proposed, with GaLore standing out as a notable example designed to reduce the memory footprint of optimizer states. However, these approaches do not alleviate the memory burden imposed by activations, rendering them unsuitable for scenarios involving long context sequences or large mini-batches. Moreover, their convergence properties are still not well-understood in the lite"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.07222","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-11T03:32:10Z","cross_cats_sorted":[],"title_canon_sha256":"c70a446d37b885c9e437c939a3f6799bf97e0dc26f4f2253d67f3e3ea53de216","abstract_canon_sha256":"18d07e6357716b6b16e2ffcfa149c82782c6cd07fe8ef97bcc35ec072b224c88"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:34.439846Z","signature_b64":"2fqlRgTYSQwXoiYKqqEyIGa1QJbsWiMx8GmhfvQ2EAaQg2MIIhV0G034VW1xytoJda/TU04dbADQp+e53Hd6Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f372af64b532b43df237dfc7534bc3dc0586a97234563d7291ef080d47cec2b6","last_reissued_at":"2026-07-05T10:12:34.439359Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:34.439359Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Memory Efficient Randomized Subspace Optimization Method for Training Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kun Yuan, Yiming Chen, Yin Liu, Yuan Zhang, Zaiwen Wen","submitted_at":"2025-02-11T03:32:10Z","abstract_excerpt":"The memory challenges associated with training Large Language Models (LLMs) have become a critical concern, particularly when using the Adam optimizer. To address this issue, numerous memory-efficient techniques have been proposed, with GaLore standing out as a notable example designed to reduce the memory footprint of optimizer states. However, these approaches do not alleviate the memory burden imposed by activations, rendering them unsuitable for scenarios involving long context sequences or large mini-batches. Moreover, their convergence properties are still not well-understood in the lite"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.07222","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.07222/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.07222","created_at":"2026-07-05T10:12:34.439420+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.07222v1","created_at":"2026-07-05T10:12:34.439420+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.07222","created_at":"2026-07-05T10:12:34.439420+00:00"},{"alias_kind":"pith_short_12","alias_value":"6NZK6ZFVGK2D","created_at":"2026-07-05T10:12:34.439420+00:00"},{"alias_kind":"pith_short_16","alias_value":"6NZK6ZFVGK2D34RX","created_at":"2026-07-05T10:12:34.439420+00:00"},{"alias_kind":"pith_short_8","alias_value":"6NZK6ZFV","created_at":"2026-07-05T10:12:34.439420+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.18993","citing_title":"CR-Net: Scaling Parameter-Efficient Training with Cross-Layer Low-Rank Structure","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12131","citing_title":"BOOST: BOttleneck-Optimized Scalable Training Framework for Low-Rank Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10288","citing_title":"BROS: Bias-Corrected Randomized Subspaces for Memory-Efficient Single-Loop Bilevel Optimization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00650","citing_title":"AdaMeZO: Adam-style Zeroth-Order Optimizer for LLM Fine-tuning Without Maintaining the Moments","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q","json":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q.json","graph_json":"https://pith.science/api/pith-number/6NZK6ZFVGK2D34RX37DVGS6D3Q/graph.json","events_json":"https://pith.science/api/pith-number/6NZK6ZFVGK2D34RX37DVGS6D3Q/events.json","paper":"https://pith.science/paper/6NZK6ZFV"},"agent_actions":{"view_html":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q","download_json":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q.json","view_paper":"https://pith.science/paper/6NZK6ZFV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.07222&json=true","fetch_graph":"https://pith.science/api/pith-number/6NZK6ZFVGK2D34RX37DVGS6D3Q/graph.json","fetch_events":"https://pith.science/api/pith-number/6NZK6ZFVGK2D34RX37DVGS6D3Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q/action/storage_attestation","attest_author":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q/action/author_attestation","sign_citation":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q/action/citation_signature","submit_replication":"https://pith.science/pith/6NZK6ZFVGK2D34RX37DVGS6D3Q/action/replication_record"}},"created_at":"2026-07-05T10:12:34.439420+00:00","updated_at":"2026-07-05T10:12:34.439420+00:00"}