{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:DPSOYAJ3QPPPFK4HAS6H7YPAOE","short_pith_number":"pith:DPSOYAJ3","schema_version":"1.0","canonical_sha256":"1be4ec013b83def2ab8704bc7fe1e071234cf18126378bc0617687aba5e18b65","source":{"kind":"arxiv","id":"1806.07572","version":4},"attestation_state":"computed","paper":{"title":"Neural Tangent Kernel: Convergence and Generalization in Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","math.PR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Arthur Jacot, Cl\\'ement Hongler, Franck Gabriel","submitted_at":"2018-06-20T06:35:46Z","abstract_excerpt":"At initialization, artificial neural networks (ANNs) are equivalent to Gaussian processes in the infinite-width limit, thus connecting them to kernel methods. We prove that the evolution of an ANN during training can also be described by a kernel: during gradient descent on the parameters of an ANN, the network function $f_\\theta$ (which maps input vectors to output vectors) follows the kernel gradient of the functional cost (which is convex, in contrast to the parameter cost) w.r.t. a new kernel: the Neural Tangent Kernel (NTK). This kernel is central to describe the generalization features o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1806.07572","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-06-20T06:35:46Z","cross_cats_sorted":["cs.NE","math.PR","stat.ML"],"title_canon_sha256":"7b8192f854a8e9c68dadf8a04d9ac0497f88578ee61f929eed6f2d45ad31d6c2","abstract_canon_sha256":"6621f9c47f305d62d63c3f9001c140ef2d7ccf46c50c205ab712e55694dee7e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:39:08.798405Z","signature_b64":"N7j6sHtJUtLHUaiJq3Hnyy/1CZel4wl7Q2kUyt0OOdWzUustIrzgA4VhAw6h68E+5GAbI1hf/ZtnF4Jca4deBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1be4ec013b83def2ab8704bc7fe1e071234cf18126378bc0617687aba5e18b65","last_reissued_at":"2026-07-05T00:39:08.797794Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:39:08.797794Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Tangent Kernel: Convergence and Generalization in Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","math.PR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Arthur Jacot, Cl\\'ement Hongler, Franck Gabriel","submitted_at":"2018-06-20T06:35:46Z","abstract_excerpt":"At initialization, artificial neural networks (ANNs) are equivalent to Gaussian processes in the infinite-width limit, thus connecting them to kernel methods. We prove that the evolution of an ANN during training can also be described by a kernel: during gradient descent on the parameters of an ANN, the network function $f_\\theta$ (which maps input vectors to output vectors) follows the kernel gradient of the functional cost (which is convex, in contrast to the parameter cost) w.r.t. a new kernel: the Neural Tangent Kernel (NTK). This kernel is central to describe the generalization features o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1806.07572","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1806.07572/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1806.07572","created_at":"2026-07-05T00:39:08.797873+00:00"},{"alias_kind":"arxiv_version","alias_value":"1806.07572v4","created_at":"2026-07-05T00:39:08.797873+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1806.07572","created_at":"2026-07-05T00:39:08.797873+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPSOYAJ3QPPP","created_at":"2026-07-05T00:39:08.797873+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPSOYAJ3QPPPFK4H","created_at":"2026-07-05T00:39:08.797873+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPSOYAJ3","created_at":"2026-07-05T00:39:08.797873+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22019","citing_title":"Channel Location Constrains the Auditability of Subliminal Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11317","citing_title":"Lectures on Semiclassical Methods for Composite Operators","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11319","citing_title":"Learning from almost nothing: How neural networks survive heavy input corruption","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09950","citing_title":"Integrating Out, Twice:The Open-System Case That Neural-Network Ensemble Theory Is Missing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08316","citing_title":"Some Inverse Problems in Particle Physics","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07247","citing_title":"Theory of learning of high-dimensional controlled non-linear dynamical systems (I): models and methods","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28585","citing_title":"Outer-Momentum Restarting in High-Dimensional Two-Phase Optimization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30860","citing_title":"Bayesian Inference with Shaped Deep Non-linear MLPs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2102.11840","citing_title":"Convergence rates for gradient descent in the training of overparameterized artificial neural networks with piecewise affine activation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2212.08989","citing_title":"Deep learning applied to computational mechanics: A comprehensive review, state of the art, and the classics","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22115","citing_title":"Physics-Informed Neural Networks with Attention Feature Expansion for Monge-Amp\\`ere Equations","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16622","citing_title":"Does Weight Decay Enhance Training Stability?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19195","citing_title":"The Thermodynamic Costs of Simple Linear Regression","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13788","citing_title":"Force-Aware Neural Tangent Kernels for Scalable and Robust Active Learning of MLIPs","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17907","citing_title":"Approximating Simple ReLU Networks based on Spectral Decomposition of Fisher Information","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2010.01412","citing_title":"Sharpness-Aware Minimization for Efficiently Improving Generalization","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13788","citing_title":"Force-Aware Neural Tangent Kernels for Scalable and Robust Active Learning of MLIPs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13183","citing_title":"Neural Networks, Dispersion Relations and the Thermal Bootstrap","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13160","citing_title":"Kernel-based guarantees for nonlinear parametric models in Bayesian optimization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2212.04089","citing_title":"Editing Models with Task Arithmetic","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06563","citing_title":"Criticality and Saturation in Orthogonal Neural Networks","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18686","citing_title":"Neural Spectral Bias and Conformal Correlators I: Introduction and Applications","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18673","citing_title":"Neural Networks Reveal a Universal Bias in Conformal Correlators","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16431","citing_title":"Dimensional Criticality at Grokking Across MLPs and Transformers","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04655","citing_title":"Grokking as Dimensional Phase Transition in Neural Networks","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE","json":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE.json","graph_json":"https://pith.science/api/pith-number/DPSOYAJ3QPPPFK4HAS6H7YPAOE/graph.json","events_json":"https://pith.science/api/pith-number/DPSOYAJ3QPPPFK4HAS6H7YPAOE/events.json","paper":"https://pith.science/paper/DPSOYAJ3"},"agent_actions":{"view_html":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE","download_json":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE.json","view_paper":"https://pith.science/paper/DPSOYAJ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1806.07572&json=true","fetch_graph":"https://pith.science/api/pith-number/DPSOYAJ3QPPPFK4HAS6H7YPAOE/graph.json","fetch_events":"https://pith.science/api/pith-number/DPSOYAJ3QPPPFK4HAS6H7YPAOE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE/action/storage_attestation","attest_author":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE/action/author_attestation","sign_citation":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE/action/citation_signature","submit_replication":"https://pith.science/pith/DPSOYAJ3QPPPFK4HAS6H7YPAOE/action/replication_record"}},"created_at":"2026-07-05T00:39:08.797873+00:00","updated_at":"2026-07-05T00:39:08.797873+00:00"}