{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MRN2RTRZS2DS5GIEK3RFIWI5KN","short_pith_number":"pith:MRN2RTRZ","schema_version":"1.0","canonical_sha256":"645ba8ce3996872e990456e254591d5366ac0a0ffbe5758cd991853f316c5867","source":{"kind":"arxiv","id":"2408.11804","version":1},"attestation_state":"computed","paper":{"title":"Approaching Deep Learning through the Spectral Dynamics of Weights","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Yunis, Gal Vardi, Karen Livescu, Kumar Kshitij Patel, Matthew R. Walter, Michael Maire, Pedro Savarese, Samuel Wheeler","submitted_at":"2024-08-21T17:48:01Z","abstract_excerpt":"We propose an empirical approach centered on the spectral dynamics of weights -- the behavior of singular values and vectors during optimization -- to unify and clarify several phenomena in deep learning. We identify a consistent bias in optimization across various experiments, from small-scale ``grokking'' to large-scale tasks like image classification with ConvNets, image generation with UNets, speech recognition with LSTMs, and language modeling with Transformers. We also demonstrate that weight decay enhances this bias beyond its role as a norm regularizer, even in practical systems. Moreo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11804","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-21T17:48:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b2d5b181cf62f0b493d10224f966fad31c8a95200510c2d8b4718508ac9e7fcc","abstract_canon_sha256":"7383ca3651c7169847f65a1b835f51cb17fdac39b9ba09c5d4112f2c7072d763"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:49.362452Z","signature_b64":"x4cRFADkqpvQliCUHJufGUWtzhgSwAImKj1GHxxRvh51pDxJy4ImeEgFKFDjASkgAcHH2rQDrhobMeQrOUNBCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"645ba8ce3996872e990456e254591d5366ac0a0ffbe5758cd991853f316c5867","last_reissued_at":"2026-07-05T08:57:49.361947Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:49.361947Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Approaching Deep Learning through the Spectral Dynamics of Weights","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Yunis, Gal Vardi, Karen Livescu, Kumar Kshitij Patel, Matthew R. Walter, Michael Maire, Pedro Savarese, Samuel Wheeler","submitted_at":"2024-08-21T17:48:01Z","abstract_excerpt":"We propose an empirical approach centered on the spectral dynamics of weights -- the behavior of singular values and vectors during optimization -- to unify and clarify several phenomena in deep learning. We identify a consistent bias in optimization across various experiments, from small-scale ``grokking'' to large-scale tasks like image classification with ConvNets, image generation with UNets, speech recognition with LSTMs, and language modeling with Transformers. We also demonstrate that weight decay enhances this bias beyond its role as a norm regularizer, even in practical systems. Moreo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11804","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11804/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11804","created_at":"2026-07-05T08:57:49.362007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11804v1","created_at":"2026-07-05T08:57:49.362007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11804","created_at":"2026-07-05T08:57:49.362007+00:00"},{"alias_kind":"pith_short_12","alias_value":"MRN2RTRZS2DS","created_at":"2026-07-05T08:57:49.362007+00:00"},{"alias_kind":"pith_short_16","alias_value":"MRN2RTRZS2DS5GIE","created_at":"2026-07-05T08:57:49.362007+00:00"},{"alias_kind":"pith_short_8","alias_value":"MRN2RTRZ","created_at":"2026-07-05T08:57:49.362007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06639","citing_title":"At-Grok Is Not Converged:A Measurement-Validity Audit for Grokking Representation Metrics","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04754","citing_title":"Beyond Structural Symmetries: Linear Mode Connectivity via Neuron Identifiability","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04405","citing_title":"Low-Rank Decay for Grokking in Scale-Invariant Transformers: A Spectral-Geometric View","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16622","citing_title":"Does Weight Decay Enhance Training Stability?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18450","citing_title":"Random Matrix Theory of Early-Stopped Gradient Flow: A Transient BBP Scenario","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN","json":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN.json","graph_json":"https://pith.science/api/pith-number/MRN2RTRZS2DS5GIEK3RFIWI5KN/graph.json","events_json":"https://pith.science/api/pith-number/MRN2RTRZS2DS5GIEK3RFIWI5KN/events.json","paper":"https://pith.science/paper/MRN2RTRZ"},"agent_actions":{"view_html":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN","download_json":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN.json","view_paper":"https://pith.science/paper/MRN2RTRZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11804&json=true","fetch_graph":"https://pith.science/api/pith-number/MRN2RTRZS2DS5GIEK3RFIWI5KN/graph.json","fetch_events":"https://pith.science/api/pith-number/MRN2RTRZS2DS5GIEK3RFIWI5KN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN/action/storage_attestation","attest_author":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN/action/author_attestation","sign_citation":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN/action/citation_signature","submit_replication":"https://pith.science/pith/MRN2RTRZS2DS5GIEK3RFIWI5KN/action/replication_record"}},"created_at":"2026-07-05T08:57:49.362007+00:00","updated_at":"2026-07-05T08:57:49.362007+00:00"}