{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YTJYAWIUCGH4KAIKRSUXRV4VV7","short_pith_number":"pith:YTJYAWIU","schema_version":"1.0","canonical_sha256":"c4d3805914118fc5010a8ca978d795afd6af63fbebe35392803287bf100ae49f","source":{"kind":"arxiv","id":"2412.13148","version":3},"attestation_state":"computed","paper":{"title":"SWAN: SGD with Normalization and Whitening Enables Stateless LLM Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chao Ma, Edward Meeds, Meyer Scetbon, Wenbo Gong","submitted_at":"2024-12-17T18:13:18Z","abstract_excerpt":"Adaptive optimizers such as Adam (Kingma & Ba, 2015) have been central to the success of large language models. However, they often require to maintain optimizer states throughout training, which can result in memory requirements several times greater than the model footprint. This overhead imposes constraints on scalability and computational efficiency. Stochastic Gradient Descent (SGD), in contrast, is a stateless optimizer, as it does not track state variables during training. Consequently, it achieves optimal memory efficiency. However, its capability in LLM training is limited (Zhao et al"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.13148","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-17T18:13:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c3f945a2a363b6b07b2b8b01ab0da424e254b311bc0566389303f2fb0b9cdb32","abstract_canon_sha256":"2cf28af657ec61d028510ce92ecc2ec04a5fe159d95a92631f968afbd189f7b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:52.347132Z","signature_b64":"Iby7c7KFPy96aaSHtVFDXgtq2Rgo7mV5rpJdS4bCi8rnF3S0SOYTM7wQo8pTaCB55cmAsa0fGgoQTLXrn6dHDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4d3805914118fc5010a8ca978d795afd6af63fbebe35392803287bf100ae49f","last_reissued_at":"2026-07-05T10:17:52.346651Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:52.346651Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SWAN: SGD with Normalization and Whitening Enables Stateless LLM Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chao Ma, Edward Meeds, Meyer Scetbon, Wenbo Gong","submitted_at":"2024-12-17T18:13:18Z","abstract_excerpt":"Adaptive optimizers such as Adam (Kingma & Ba, 2015) have been central to the success of large language models. However, they often require to maintain optimizer states throughout training, which can result in memory requirements several times greater than the model footprint. This overhead imposes constraints on scalability and computational efficiency. Stochastic Gradient Descent (SGD), in contrast, is a stateless optimizer, as it does not track state variables during training. Consequently, it achieves optimal memory efficiency. However, its capability in LLM training is limited (Zhao et al"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.13148","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.13148/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.13148","created_at":"2026-07-05T10:17:52.346710+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.13148v3","created_at":"2026-07-05T10:17:52.346710+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.13148","created_at":"2026-07-05T10:17:52.346710+00:00"},{"alias_kind":"pith_short_12","alias_value":"YTJYAWIUCGH4","created_at":"2026-07-05T10:17:52.346710+00:00"},{"alias_kind":"pith_short_16","alias_value":"YTJYAWIUCGH4KAIK","created_at":"2026-07-05T10:17:52.346710+00:00"},{"alias_kind":"pith_short_8","alias_value":"YTJYAWIU","created_at":"2026-07-05T10:17:52.346710+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27216","citing_title":"Hierarchical Muon: Tiled Newton-Schulz Updates for Efficient Muon Optimization","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16659","citing_title":"Memory-Efficient LLM Pretraining via Minimalist Optimizer Design","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04418","citing_title":"Demystifying Manifold Constraints in LLM Pre-training","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7","json":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7.json","graph_json":"https://pith.science/api/pith-number/YTJYAWIUCGH4KAIKRSUXRV4VV7/graph.json","events_json":"https://pith.science/api/pith-number/YTJYAWIUCGH4KAIKRSUXRV4VV7/events.json","paper":"https://pith.science/paper/YTJYAWIU"},"agent_actions":{"view_html":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7","download_json":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7.json","view_paper":"https://pith.science/paper/YTJYAWIU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.13148&json=true","fetch_graph":"https://pith.science/api/pith-number/YTJYAWIUCGH4KAIKRSUXRV4VV7/graph.json","fetch_events":"https://pith.science/api/pith-number/YTJYAWIUCGH4KAIKRSUXRV4VV7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7/action/storage_attestation","attest_author":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7/action/author_attestation","sign_citation":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7/action/citation_signature","submit_replication":"https://pith.science/pith/YTJYAWIUCGH4KAIKRSUXRV4VV7/action/replication_record"}},"created_at":"2026-07-05T10:17:52.346710+00:00","updated_at":"2026-07-05T10:17:52.346710+00:00"}