{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XWFK5U3QI5SBGN2ILXPKEEZ45V","short_pith_number":"pith:XWFK5U3Q","schema_version":"1.0","canonical_sha256":"bd8aaed37047641337485ddea2133ced4f9198fd66fa48d4756da6b4b3976001","source":{"kind":"arxiv","id":"2403.08160","version":1},"attestation_state":"computed","paper":{"title":"Asymptotics of Random Feature Regression Beyond the Linear Scaling Regime","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Hong Hu, Theodor Misiakiewicz, Yue M. Lu","submitted_at":"2024-03-13T00:59:25Z","abstract_excerpt":"Recent advances in machine learning have been achieved by using overparametrized models trained until near interpolation of the training data. It was shown, e.g., through the double descent phenomenon, that the number of parameters is a poor proxy for the model complexity and generalization capabilities. This leaves open the question of understanding the impact of parametrization on the performance of these models. How does model complexity and generalization depend on the number of parameters $p$? How should we choose $p$ relative to the sample size $n$ to achieve optimal test error?\n  In thi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08160","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2024-03-13T00:59:25Z","cross_cats_sorted":["cs.LG","math.ST","stat.TH"],"title_canon_sha256":"ed235cb4a8895a4e8fd067c079c625ddf0e1ca90d0fb14add868790f14f57638","abstract_canon_sha256":"6d330c08d70e74f3608e5d37cf8d9244f104325596700350a3488fc625ce017b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:34.910584Z","signature_b64":"2ULf3JkCchapyVWaSePhvGJlag5WSQzS2gtm45Xx3KH/2QTDiCA9eTNrBOKBtgumi4b62GHv72eqs3xFN/OcBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bd8aaed37047641337485ddea2133ced4f9198fd66fa48d4756da6b4b3976001","last_reissued_at":"2026-07-05T07:55:34.910150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:34.910150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Asymptotics of Random Feature Regression Beyond the Linear Scaling Regime","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Hong Hu, Theodor Misiakiewicz, Yue M. Lu","submitted_at":"2024-03-13T00:59:25Z","abstract_excerpt":"Recent advances in machine learning have been achieved by using overparametrized models trained until near interpolation of the training data. It was shown, e.g., through the double descent phenomenon, that the number of parameters is a poor proxy for the model complexity and generalization capabilities. This leaves open the question of understanding the impact of parametrization on the performance of these models. How does model complexity and generalization depend on the number of parameters $p$? How should we choose $p$ relative to the sample size $n$ to achieve optimal test error?\n  In thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08160","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08160/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08160","created_at":"2026-07-05T07:55:34.910206+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08160v1","created_at":"2026-07-05T07:55:34.910206+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08160","created_at":"2026-07-05T07:55:34.910206+00:00"},{"alias_kind":"pith_short_12","alias_value":"XWFK5U3QI5SB","created_at":"2026-07-05T07:55:34.910206+00:00"},{"alias_kind":"pith_short_16","alias_value":"XWFK5U3QI5SBGN2I","created_at":"2026-07-05T07:55:34.910206+00:00"},{"alias_kind":"pith_short_8","alias_value":"XWFK5U3Q","created_at":"2026-07-05T07:55:34.910206+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14567","citing_title":"Scaling Laws from Sequential Feature Recovery: A Solvable Hierarchical Model","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V","json":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V.json","graph_json":"https://pith.science/api/pith-number/XWFK5U3QI5SBGN2ILXPKEEZ45V/graph.json","events_json":"https://pith.science/api/pith-number/XWFK5U3QI5SBGN2ILXPKEEZ45V/events.json","paper":"https://pith.science/paper/XWFK5U3Q"},"agent_actions":{"view_html":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V","download_json":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V.json","view_paper":"https://pith.science/paper/XWFK5U3Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08160&json=true","fetch_graph":"https://pith.science/api/pith-number/XWFK5U3QI5SBGN2ILXPKEEZ45V/graph.json","fetch_events":"https://pith.science/api/pith-number/XWFK5U3QI5SBGN2ILXPKEEZ45V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V/action/storage_attestation","attest_author":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V/action/author_attestation","sign_citation":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V/action/citation_signature","submit_replication":"https://pith.science/pith/XWFK5U3QI5SBGN2ILXPKEEZ45V/action/replication_record"}},"created_at":"2026-07-05T07:55:34.910206+00:00","updated_at":"2026-07-05T07:55:34.910206+00:00"}