{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:OFE23OFTMB6ZASMXZBJFZVPL7P","short_pith_number":"pith:OFE23OFT","schema_version":"1.0","canonical_sha256":"7149adb8b3607d904997c8525cd5ebfbee89093f4ca4602b0f306236c316218a","source":{"kind":"arxiv","id":"2205.15809","version":2},"attestation_state":"computed","paper":{"title":"Feature Learning in $L_{2}$-regularized DNNs: Attraction/Repulsion and Sparsity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.NE"],"primary_cat":"stat.ML","authors_text":"Arthur Jacot, Cl\\'ement Hongler, Eugene Golikov, Franck Gabriel","submitted_at":"2022-05-31T14:10:15Z","abstract_excerpt":"We study the loss surface of DNNs with $L_{2}$ regularization. We show that the loss in terms of the parameters can be reformulated into a loss in terms of the layerwise activations $Z_{\\ell}$ of the training set. This reformulation reveals the dynamics behind feature learning: each hidden representations $Z_{\\ell}$ are optimal w.r.t. to an attraction/repulsion problem and interpolate between the input and output representations, keeping as little information from the input as necessary to construct the activation of the next layer. For positively homogeneous non-linearities, the loss can be f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.15809","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2022-05-31T14:10:15Z","cross_cats_sorted":["cs.AI","cs.LG","cs.NE"],"title_canon_sha256":"3623b979d3879a16479707b7de50515661265e0f15d413f516372ce2c404f1bc","abstract_canon_sha256":"505721df2f5140dd0cdd7d281379b21fdd83f66a01ba7b6a55dbec3228c76046"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:31.965098Z","signature_b64":"nHhONZyXbg+gxUjlJ7qO6o+qhMBfNYo2nP4hv+eS+V0CleuVgnV4R6hrpLM3Fqz2QUoa8Xv+vSXdpSuJQKUsCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7149adb8b3607d904997c8525cd5ebfbee89093f4ca4602b0f306236c316218a","last_reissued_at":"2026-07-05T05:06:31.964637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:31.964637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feature Learning in $L_{2}$-regularized DNNs: Attraction/Repulsion and Sparsity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.NE"],"primary_cat":"stat.ML","authors_text":"Arthur Jacot, Cl\\'ement Hongler, Eugene Golikov, Franck Gabriel","submitted_at":"2022-05-31T14:10:15Z","abstract_excerpt":"We study the loss surface of DNNs with $L_{2}$ regularization. We show that the loss in terms of the parameters can be reformulated into a loss in terms of the layerwise activations $Z_{\\ell}$ of the training set. This reformulation reveals the dynamics behind feature learning: each hidden representations $Z_{\\ell}$ are optimal w.r.t. to an attraction/repulsion problem and interpolate between the input and output representations, keeping as little information from the input as necessary to construct the activation of the next layer. For positively homogeneous non-linearities, the loss can be f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.15809","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.15809/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.15809","created_at":"2026-07-05T05:06:31.964694+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.15809v2","created_at":"2026-07-05T05:06:31.964694+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.15809","created_at":"2026-07-05T05:06:31.964694+00:00"},{"alias_kind":"pith_short_12","alias_value":"OFE23OFTMB6Z","created_at":"2026-07-05T05:06:31.964694+00:00"},{"alias_kind":"pith_short_16","alias_value":"OFE23OFTMB6ZASMX","created_at":"2026-07-05T05:06:31.964694+00:00"},{"alias_kind":"pith_short_8","alias_value":"OFE23OFT","created_at":"2026-07-05T05:06:31.964694+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25134","citing_title":"Theoretical Analysis of Sparse Optimization with Reparameterization, Weight Decay, and Adaptive Learning Rate","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10237","citing_title":"The Benefits of Temporal Correlations: SGD Learns k-Juntas from Random Walks Efficiently","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P","json":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P.json","graph_json":"https://pith.science/api/pith-number/OFE23OFTMB6ZASMXZBJFZVPL7P/graph.json","events_json":"https://pith.science/api/pith-number/OFE23OFTMB6ZASMXZBJFZVPL7P/events.json","paper":"https://pith.science/paper/OFE23OFT"},"agent_actions":{"view_html":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P","download_json":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P.json","view_paper":"https://pith.science/paper/OFE23OFT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.15809&json=true","fetch_graph":"https://pith.science/api/pith-number/OFE23OFTMB6ZASMXZBJFZVPL7P/graph.json","fetch_events":"https://pith.science/api/pith-number/OFE23OFTMB6ZASMXZBJFZVPL7P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P/action/storage_attestation","attest_author":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P/action/author_attestation","sign_citation":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P/action/citation_signature","submit_replication":"https://pith.science/pith/OFE23OFTMB6ZASMXZBJFZVPL7P/action/replication_record"}},"created_at":"2026-07-05T05:06:31.964694+00:00","updated_at":"2026-07-05T05:06:31.964694+00:00"}