{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:XW4CDFF2NYDULOMDNH2RCWRDMI","short_pith_number":"pith:XW4CDFF2","schema_version":"1.0","canonical_sha256":"bdb82194ba6e0745b98369f5115a23623bb86e2a2f044e534c30fe253501dc2f","source":{"kind":"arxiv","id":"2208.02789","version":1},"attestation_state":"computed","paper":{"title":"Feature selection with gradient descent on two-layer networks in low-rotation regimes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Matus Telgarsky","submitted_at":"2022-08-04T17:43:36Z","abstract_excerpt":"This work establishes low test error of gradient flow (GF) and stochastic gradient descent (SGD) on two-layer ReLU networks with standard initialization, in three regimes where key sets of weights rotate little (either naturally due to GF and SGD, or due to an artificial constraint), and making use of margins as the core analytic technique. The first regime is near initialization, specifically until the weights have moved by $\\mathcal{O}(\\sqrt m)$, where $m$ denotes the network width, which is in sharp contrast to the $\\mathcal{O}(1)$ weight motion allowed by the Neural Tangent Kernel (NTK); h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.02789","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-08-04T17:43:36Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"4086da379451db52c0cfa001011ab0e7b14e11a507a403e3ac7197064e17a5ed","abstract_canon_sha256":"7f0cc7db414db94bfc1a20683ceccc3a23f3572b0f46471198497dff572517f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:46:08.981296Z","signature_b64":"nfIzEjhlWWbAT79NFvOFgkR0Tp48DMCmRkOnYBo8xEAwdfQEZ2FJYD0yYmSuBRBoqCuXPvxzq9zcrpkginCfAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bdb82194ba6e0745b98369f5115a23623bb86e2a2f044e534c30fe253501dc2f","last_reissued_at":"2026-07-05T04:46:08.980899Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:46:08.980899Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feature selection with gradient descent on two-layer networks in low-rotation regimes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Matus Telgarsky","submitted_at":"2022-08-04T17:43:36Z","abstract_excerpt":"This work establishes low test error of gradient flow (GF) and stochastic gradient descent (SGD) on two-layer ReLU networks with standard initialization, in three regimes where key sets of weights rotate little (either naturally due to GF and SGD, or due to an artificial constraint), and making use of margins as the core analytic technique. The first regime is near initialization, specifically until the weights have moved by $\\mathcal{O}(\\sqrt m)$, where $m$ denotes the network width, which is in sharp contrast to the $\\mathcal{O}(1)$ weight motion allowed by the Neural Tangent Kernel (NTK); h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.02789","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.02789/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.02789","created_at":"2026-07-05T04:46:08.980951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.02789v1","created_at":"2026-07-05T04:46:08.980951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.02789","created_at":"2026-07-05T04:46:08.980951+00:00"},{"alias_kind":"pith_short_12","alias_value":"XW4CDFF2NYDU","created_at":"2026-07-05T04:46:08.980951+00:00"},{"alias_kind":"pith_short_16","alias_value":"XW4CDFF2NYDULOMD","created_at":"2026-07-05T04:46:08.980951+00:00"},{"alias_kind":"pith_short_8","alias_value":"XW4CDFF2","created_at":"2026-07-05T04:46:08.980951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23364","citing_title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","ref_index":121,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI","json":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI.json","graph_json":"https://pith.science/api/pith-number/XW4CDFF2NYDULOMDNH2RCWRDMI/graph.json","events_json":"https://pith.science/api/pith-number/XW4CDFF2NYDULOMDNH2RCWRDMI/events.json","paper":"https://pith.science/paper/XW4CDFF2"},"agent_actions":{"view_html":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI","download_json":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI.json","view_paper":"https://pith.science/paper/XW4CDFF2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.02789&json=true","fetch_graph":"https://pith.science/api/pith-number/XW4CDFF2NYDULOMDNH2RCWRDMI/graph.json","fetch_events":"https://pith.science/api/pith-number/XW4CDFF2NYDULOMDNH2RCWRDMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI/action/storage_attestation","attest_author":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI/action/author_attestation","sign_citation":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI/action/citation_signature","submit_replication":"https://pith.science/pith/XW4CDFF2NYDULOMDNH2RCWRDMI/action/replication_record"}},"created_at":"2026-07-05T04:46:08.980951+00:00","updated_at":"2026-07-05T04:46:08.980951+00:00"}