{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:D2HSZHQ4QAW6XGOAV7VH6LTK6L","short_pith_number":"pith:D2HSZHQ4","schema_version":"1.0","canonical_sha256":"1e8f2c9e1c802deb99c0afea7f2e6af2e9ce84b89a08aca32e5da3effefcc1d3","source":{"kind":"arxiv","id":"2206.05794","version":7},"attestation_state":"computed","paper":{"title":"SGD and Weight Decay Secretly Minimize the Rank of Your Neural Network","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aparna Gupte, Tomaso Poggio, Tomer Galanti, Zachary S. Siegel","submitted_at":"2022-06-12T17:06:35Z","abstract_excerpt":"We investigate the inherent bias of Stochastic Gradient Descent (SGD) toward learning low-rank weight matrices during the training of deep neural networks. Our results demonstrate that training with mini-batch SGD and weight decay induces a bias toward rank minimization in the weight matrices. Specifically, we show both theoretically and empirically that this bias becomes more pronounced with smaller batch sizes, higher learning rates, or stronger weight decay. Additionally, we predict and empirically confirm that weight decay is essential for this bias to occur. Unlike previous literature, ou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.05794","kind":"arxiv","version":7},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-12T17:06:35Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"63774584c42548eefae390f02839fd2029f9e1050d17d086d69cde7d26049cf4","abstract_canon_sha256":"591d5f67b26456dd05dd7e9088212b72d1cf48c2872e1a796532bc40af12767f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:35.833808Z","signature_b64":"hN67EXi+tYkSJ+woj03sOrrecUNYhrFVigVQGqw1S9B4Xgs28HuVFpb2OF/7UrzHSCeMbyeMEBXkrINxrJRXDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e8f2c9e1c802deb99c0afea7f2e6af2e9ce84b89a08aca32e5da3effefcc1d3","last_reissued_at":"2026-07-05T09:22:35.833272Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:35.833272Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SGD and Weight Decay Secretly Minimize the Rank of Your Neural Network","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aparna Gupte, Tomaso Poggio, Tomer Galanti, Zachary S. Siegel","submitted_at":"2022-06-12T17:06:35Z","abstract_excerpt":"We investigate the inherent bias of Stochastic Gradient Descent (SGD) toward learning low-rank weight matrices during the training of deep neural networks. Our results demonstrate that training with mini-batch SGD and weight decay induces a bias toward rank minimization in the weight matrices. Specifically, we show both theoretically and empirically that this bias becomes more pronounced with smaller batch sizes, higher learning rates, or stronger weight decay. Additionally, we predict and empirically confirm that weight decay is essential for this bias to occur. Unlike previous literature, ou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.05794","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.05794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.05794","created_at":"2026-07-05T09:22:35.833349+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.05794v7","created_at":"2026-07-05T09:22:35.833349+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.05794","created_at":"2026-07-05T09:22:35.833349+00:00"},{"alias_kind":"pith_short_12","alias_value":"D2HSZHQ4QAW6","created_at":"2026-07-05T09:22:35.833349+00:00"},{"alias_kind":"pith_short_16","alias_value":"D2HSZHQ4QAW6XGOA","created_at":"2026-07-05T09:22:35.833349+00:00"},{"alias_kind":"pith_short_8","alias_value":"D2HSZHQ4","created_at":"2026-07-05T09:22:35.833349+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05863","citing_title":"Deciphering Two Training Clocks in Grokking via Deep Linear Network Theory with Conditional ReLU Reduction","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23087","citing_title":"The Implicit Bias of Depth: From Neural Collapse to Softmax Codes","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20441","citing_title":"Weight Decay Regimes in Grokking Transformers: Cheap Online Diagnostics","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16622","citing_title":"Does Weight Decay Enhance Training Stability?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03473","citing_title":"Evolutionary Search for Automated Design of Uncertainty Quantification Methods","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L","json":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L.json","graph_json":"https://pith.science/api/pith-number/D2HSZHQ4QAW6XGOAV7VH6LTK6L/graph.json","events_json":"https://pith.science/api/pith-number/D2HSZHQ4QAW6XGOAV7VH6LTK6L/events.json","paper":"https://pith.science/paper/D2HSZHQ4"},"agent_actions":{"view_html":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L","download_json":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L.json","view_paper":"https://pith.science/paper/D2HSZHQ4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.05794&json=true","fetch_graph":"https://pith.science/api/pith-number/D2HSZHQ4QAW6XGOAV7VH6LTK6L/graph.json","fetch_events":"https://pith.science/api/pith-number/D2HSZHQ4QAW6XGOAV7VH6LTK6L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L/action/storage_attestation","attest_author":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L/action/author_attestation","sign_citation":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L/action/citation_signature","submit_replication":"https://pith.science/pith/D2HSZHQ4QAW6XGOAV7VH6LTK6L/action/replication_record"}},"created_at":"2026-07-05T09:22:35.833349+00:00","updated_at":"2026-07-05T09:22:35.833349+00:00"}