{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:KM3XVFBLGQH4REMSH2PKSYAUYG","short_pith_number":"pith:KM3XVFBL","schema_version":"1.0","canonical_sha256":"53377a942b340fc891923e9ea96014c18b3e070c04d48404622a9fc0f5e4782a","source":{"kind":"arxiv","id":"2011.14522","version":3},"attestation_state":"computed","paper":{"title":"Feature Learning in Infinite-Width Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.NE"],"primary_cat":"cs.LG","authors_text":"Edward J. Hu, Greg Yang","submitted_at":"2020-11-30T03:21:05Z","abstract_excerpt":"As its width tends to infinity, a deep neural network's behavior under gradient descent can become simplified and predictable (e.g. given by the Neural Tangent Kernel (NTK)), if it is parametrized appropriately (e.g. the NTK parametrization). However, we show that the standard and NTK parametrizations of a neural network do not admit infinite-width limits that can learn features, which is crucial for pretraining and transfer learning such as with BERT. We propose simple modifications to the standard parametrization to allow for feature learning in the limit. Using the *Tensor Programs* techniq"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.14522","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-11-30T03:21:05Z","cross_cats_sorted":["cond-mat.dis-nn","cs.NE"],"title_canon_sha256":"af3f63d420af4ec0ab564348dee3607a0fdc5a634a914b86481d94bf91f39afd","abstract_canon_sha256":"c45b348ed53c2cbef2c78c061c9521ea933eeb362fbecfc473e6a5d533064ee8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:40:22.744946Z","signature_b64":"TXHgPMy+OLg7hsdfGBXIcjfv7BGrIw9aX96wh19ZdzGaPpAfzFUfvhZTrIvcrtsDJzeICf1YD77fIQy8P5wwCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53377a942b340fc891923e9ea96014c18b3e070c04d48404622a9fc0f5e4782a","last_reissued_at":"2026-07-05T04:40:22.744476Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:40:22.744476Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feature Learning in Infinite-Width Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.NE"],"primary_cat":"cs.LG","authors_text":"Edward J. Hu, Greg Yang","submitted_at":"2020-11-30T03:21:05Z","abstract_excerpt":"As its width tends to infinity, a deep neural network's behavior under gradient descent can become simplified and predictable (e.g. given by the Neural Tangent Kernel (NTK)), if it is parametrized appropriately (e.g. the NTK parametrization). However, we show that the standard and NTK parametrizations of a neural network do not admit infinite-width limits that can learn features, which is crucial for pretraining and transfer learning such as with BERT. We propose simple modifications to the standard parametrization to allow for feature learning in the limit. Using the *Tensor Programs* techniq"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.14522","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.14522/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.14522","created_at":"2026-07-05T04:40:22.744532+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.14522v3","created_at":"2026-07-05T04:40:22.744532+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.14522","created_at":"2026-07-05T04:40:22.744532+00:00"},{"alias_kind":"pith_short_12","alias_value":"KM3XVFBLGQH4","created_at":"2026-07-05T04:40:22.744532+00:00"},{"alias_kind":"pith_short_16","alias_value":"KM3XVFBLGQH4REMS","created_at":"2026-07-05T04:40:22.744532+00:00"},{"alias_kind":"pith_short_8","alias_value":"KM3XVFBL","created_at":"2026-07-05T04:40:22.744532+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23364","citing_title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22019","citing_title":"Channel Location Constrains the Auditability of Subliminal Learning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21645","citing_title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":283,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30831","citing_title":"Geometric Dyson Brownian Motions and the Free Log-Normal Limit for a Non-Square Gaussian Matrix Product","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22297","citing_title":"One LR Doesn't Fit All: Heavy-Tail Guided Layerwise Learning Rates for LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22297","citing_title":"One LR Doesn't Fit All: Heavy-Tail Guided Layerwise Learning Rates for LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2505.24333","citing_title":"Two failure modes of deep transformers and how to avoid them: a unified theory of signal propagation at initialisation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11170","citing_title":"Unlearning with Asymmetric Sources: Improved Unlearning-Utility Trade-off with Public Data","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12492","citing_title":"Pion: A Spectrum-Preserving Optimizer via Orthogonal Equivalence Transformation","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11506","citing_title":"Principled Design of Diffusion-based Optimizers for Inverse Problems","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2303.10512","citing_title":"AdaLoRA: Adaptive Budget Allocation for Parameter-Efficient Fine-Tuning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21352","citing_title":"CARE: Counselor-Aligned Response Engine for Online Mental-Health Support","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2106.09685","citing_title":"LoRA: Low-Rank Adaptation of Large Language Models","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG","json":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG.json","graph_json":"https://pith.science/api/pith-number/KM3XVFBLGQH4REMSH2PKSYAUYG/graph.json","events_json":"https://pith.science/api/pith-number/KM3XVFBLGQH4REMSH2PKSYAUYG/events.json","paper":"https://pith.science/paper/KM3XVFBL"},"agent_actions":{"view_html":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG","download_json":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG.json","view_paper":"https://pith.science/paper/KM3XVFBL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.14522&json=true","fetch_graph":"https://pith.science/api/pith-number/KM3XVFBLGQH4REMSH2PKSYAUYG/graph.json","fetch_events":"https://pith.science/api/pith-number/KM3XVFBLGQH4REMSH2PKSYAUYG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG/action/storage_attestation","attest_author":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG/action/author_attestation","sign_citation":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG/action/citation_signature","submit_replication":"https://pith.science/pith/KM3XVFBLGQH4REMSH2PKSYAUYG/action/replication_record"}},"created_at":"2026-07-05T04:40:22.744532+00:00","updated_at":"2026-07-05T04:40:22.744532+00:00"}