{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AU4PGITFTTI3X2B5MNPLF2IRAP","short_pith_number":"pith:AU4PGITF","schema_version":"1.0","canonical_sha256":"0538f322659cd1bbe83d635eb2e91103f798cf09b56a87ccd5e8d3741f3d4a7d","source":{"kind":"arxiv","id":"2402.04980","version":2},"attestation_state":"computed","paper":{"title":"Asymptotics of feature learning in two-layer networks after one gradient-step","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.LG"],"primary_cat":"stat.ML","authors_text":"Bruno Loureiro, Florent Krzakala, Hugo Cui, Lenka Zdeborov\\'a, Luca Pesce, Yatin Dandi, Yue M. Lu","submitted_at":"2024-02-07T15:57:30Z","abstract_excerpt":"In this manuscript, we investigate the problem of how two-layer neural networks learn features from data, and improve over the kernel regime, after being trained with a single gradient descent step. Leveraging the insight from (Ba et al., 2022), we model the trained network by a spiked Random Features (sRF) model. Further building on recent progress on Gaussian universality (Dandi et al., 2023), we provide an exact asymptotic description of the generalization error of the sRF in the high-dimensional limit where the number of samples, the width, and the input dimension grow at a proportional ra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04980","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2024-02-07T15:57:30Z","cross_cats_sorted":["cond-mat.dis-nn","cs.LG"],"title_canon_sha256":"f0d952aa24e423795b9082def60e696328c870db7619d4a889c09eef0996a24c","abstract_canon_sha256":"b247e4d19b94a55e3c9ee6fc05df0f972f2a39e3eb9a54727a8acf5bb53c681f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:03:18.352076Z","signature_b64":"qgThJc9GppTijOl3VEkWdy9A9G/cS1m55RQOr983S47ngY0rifH9uAU7bJhu7vWwb8CqGyuPqJ8GuisleWJyDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0538f322659cd1bbe83d635eb2e91103f798cf09b56a87ccd5e8d3741f3d4a7d","last_reissued_at":"2026-07-05T09:03:18.351592Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:03:18.351592Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Asymptotics of feature learning in two-layer networks after one gradient-step","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.LG"],"primary_cat":"stat.ML","authors_text":"Bruno Loureiro, Florent Krzakala, Hugo Cui, Lenka Zdeborov\\'a, Luca Pesce, Yatin Dandi, Yue M. Lu","submitted_at":"2024-02-07T15:57:30Z","abstract_excerpt":"In this manuscript, we investigate the problem of how two-layer neural networks learn features from data, and improve over the kernel regime, after being trained with a single gradient descent step. Leveraging the insight from (Ba et al., 2022), we model the trained network by a spiked Random Features (sRF) model. Further building on recent progress on Gaussian universality (Dandi et al., 2023), we provide an exact asymptotic description of the generalization error of the sRF in the high-dimensional limit where the number of samples, the width, and the input dimension grow at a proportional ra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04980","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04980/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04980","created_at":"2026-07-05T09:03:18.351650+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04980v2","created_at":"2026-07-05T09:03:18.351650+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04980","created_at":"2026-07-05T09:03:18.351650+00:00"},{"alias_kind":"pith_short_12","alias_value":"AU4PGITFTTI3","created_at":"2026-07-05T09:03:18.351650+00:00"},{"alias_kind":"pith_short_16","alias_value":"AU4PGITFTTI3X2B5","created_at":"2026-07-05T09:03:18.351650+00:00"},{"alias_kind":"pith_short_8","alias_value":"AU4PGITF","created_at":"2026-07-05T09:03:18.351650+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07870","citing_title":"Spectral Dynamics in Deep Networks: Feature Learning, Outlier Escape, and Learning Rate Transfer","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21691","citing_title":"There Will Be a Scientific Theory of Deep Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07870","citing_title":"Spectral Dynamics in Deep Networks: Feature Learning, Outlier Escape, and Learning Rate Transfer","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP","json":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP.json","graph_json":"https://pith.science/api/pith-number/AU4PGITFTTI3X2B5MNPLF2IRAP/graph.json","events_json":"https://pith.science/api/pith-number/AU4PGITFTTI3X2B5MNPLF2IRAP/events.json","paper":"https://pith.science/paper/AU4PGITF"},"agent_actions":{"view_html":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP","download_json":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP.json","view_paper":"https://pith.science/paper/AU4PGITF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04980&json=true","fetch_graph":"https://pith.science/api/pith-number/AU4PGITFTTI3X2B5MNPLF2IRAP/graph.json","fetch_events":"https://pith.science/api/pith-number/AU4PGITFTTI3X2B5MNPLF2IRAP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP/action/storage_attestation","attest_author":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP/action/author_attestation","sign_citation":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP/action/citation_signature","submit_replication":"https://pith.science/pith/AU4PGITFTTI3X2B5MNPLF2IRAP/action/replication_record"}},"created_at":"2026-07-05T09:03:18.351650+00:00","updated_at":"2026-07-05T09:03:18.351650+00:00"}