{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WOOKY7V5BJOB4RIWY7BHVSVJWT","short_pith_number":"pith:WOOKY7V5","schema_version":"1.0","canonical_sha256":"b39cac7ebd0a5c1e4516c7c27acaa9b4f57d19aae10b201b157abe645055c55a","source":{"kind":"arxiv","id":"2510.24616","version":4},"attestation_state":"computed","paper":{"title":"Statistical physics of deep learning: Optimal learning of a multi-layer perceptron near interpolation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cond-mat.dis-nn","cond-mat.stat-mech","cs.IT","cs.LG","math.IT"],"primary_cat":"stat.ML","authors_text":"Francesco Camilli, Jean Barbier, Mauro Pastore, Minh-Toan Nguyen, Rudy Skerk","submitted_at":"2025-10-28T16:44:34Z","abstract_excerpt":"For four decades statistical physics has been providing a framework to analyse neural networks. A long-standing question remained on its capacity to tackle deep learning models capturing rich feature learning effects, thus going beyond the narrow networks or kernel methods analysed until now. We positively answer through the study of the supervised learning of a multi-layer perceptron. Importantly, (i) its width scales as the input dimension, making it more prone to feature learning than ultra wide networks, and more expressive than narrow ones or ones with fixed embedding layers; and (ii) we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.24616","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2025-10-28T16:44:34Z","cross_cats_sorted":["cond-mat.dis-nn","cond-mat.stat-mech","cs.IT","cs.LG","math.IT"],"title_canon_sha256":"c76c1bbc96eb6dd8060e7d22e65a575dc2f1862440a190f3e2cb9b2290182bdd","abstract_canon_sha256":"a51ae2281b735c03f3cf5ccccb8e4d3c323b66e76e90c10c5653c84cede1b9b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T01:24:04.510333Z","signature_b64":"KaT9vklOzRVsMZOXBwAEkH/VZnmXsh3Wl6WqQE29H2eCmjHFBpzyMeNB85uzLE9lKERuwCuxUenoD4gpWgIzAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b39cac7ebd0a5c1e4516c7c27acaa9b4f57d19aae10b201b157abe645055c55a","last_reissued_at":"2026-07-24T01:24:04.509352Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T01:24:04.509352Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Statistical physics of deep learning: Optimal learning of a multi-layer perceptron near interpolation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cond-mat.dis-nn","cond-mat.stat-mech","cs.IT","cs.LG","math.IT"],"primary_cat":"stat.ML","authors_text":"Francesco Camilli, Jean Barbier, Mauro Pastore, Minh-Toan Nguyen, Rudy Skerk","submitted_at":"2025-10-28T16:44:34Z","abstract_excerpt":"For four decades statistical physics has been providing a framework to analyse neural networks. A long-standing question remained on its capacity to tackle deep learning models capturing rich feature learning effects, thus going beyond the narrow networks or kernel methods analysed until now. We positively answer through the study of the supervised learning of a multi-layer perceptron. Importantly, (i) its width scales as the input dimension, making it more prone to feature learning than ultra wide networks, and more expressive than narrow ones or ones with fixed embedding layers; and (ii) we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.24616","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.24616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.24616","created_at":"2026-07-24T01:24:04.509811+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.24616v4","created_at":"2026-07-24T01:24:04.509811+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.24616","created_at":"2026-07-24T01:24:04.509811+00:00"},{"alias_kind":"pith_short_12","alias_value":"WOOKY7V5BJOB","created_at":"2026-07-24T01:24:04.509811+00:00"},{"alias_kind":"pith_short_16","alias_value":"WOOKY7V5BJOB4RIW","created_at":"2026-07-24T01:24:04.509811+00:00"},{"alias_kind":"pith_short_8","alias_value":"WOOKY7V5","created_at":"2026-07-24T01:24:04.509811+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":4,"sample":[{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":111,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26973","citing_title":"Signal-to-Noise Ratio and Sample Size Govern Representational Alignment in Neural Networks","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23591","citing_title":"Asymmetric Scaling Laws from Sparse Features","ref_index":71,"is_internal_anchor":true},{"citing_arxiv_id":"2604.21691","citing_title":"There Will Be a Scientific Theory of Deep Learning","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT","json":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT.json","graph_json":"https://pith.science/api/pith-number/WOOKY7V5BJOB4RIWY7BHVSVJWT/graph.json","events_json":"https://pith.science/api/pith-number/WOOKY7V5BJOB4RIWY7BHVSVJWT/events.json","paper":"https://pith.science/paper/WOOKY7V5"},"agent_actions":{"view_html":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT","download_json":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT.json","view_paper":"https://pith.science/paper/WOOKY7V5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.24616&json=true","fetch_graph":"https://pith.science/api/pith-number/WOOKY7V5BJOB4RIWY7BHVSVJWT/graph.json","fetch_events":"https://pith.science/api/pith-number/WOOKY7V5BJOB4RIWY7BHVSVJWT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT/action/storage_attestation","attest_author":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT/action/author_attestation","sign_citation":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT/action/citation_signature","submit_replication":"https://pith.science/pith/WOOKY7V5BJOB4RIWY7BHVSVJWT/action/replication_record"}},"created_at":"2026-07-24T01:24:04.509811+00:00","updated_at":"2026-07-24T01:24:04.509811+00:00"}