{"as_of":"2026-08-18T10:58:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:93c01db73c99de82dbafa70eea1e979aa9c9101d996c4f36a5f0c9b4982d0f3f","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:17:37.261961Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.19680/citation-record","integrity":"/paper/2507.19680/integrity","json":"/paper/2507.19680/citation-record.json","paper":"/paper/2507.19680"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2202.08658","last_updated":"2024-08-26T23:24:52Z","snapshot_observed_at":"2026-08-16T17:20:34.346429Z","submitted_at":"2022-02-17T13:43:06Z","title":"The merged-staircase property: a necessary and nearly sufficient condition for SGD learning of sparse functions on two-layer neural networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.08658","snapshot_observed_at":"2026-08-06T14:17:37.001723Z","title":"The merged-staircase property: a necessary and nearly sufficient condition for sgd learning of sparse functions on two-layer neural networks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.001723Z"},"links":{"cited_paper":"/paper/2202.08658","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:0f8d0bd5523828dc4334ea3d9b1e8d0c8c42151a703fd8fa62aa27fa778f0d02","observation_id":"a0e59c41-770d-4f7d-951f-88d18a6a2832","resolution":{"observed_at":"2026-08-06T14:17:37.001723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.006053Z","title":"Aiudi, R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.006053Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:2e41afa09b6c5240c2c80d9c11c56049f1a051f9ceb4927d51437934ef6b79a4","observation_id":"f23fbd5f-4ecc-495d-830b-22d5bfef3a06","resolution":{"observed_at":"2026-08-06T14:17:37.006053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14818","last_updated":"2022-06-06T14:29:46Z","snapshot_observed_at":"2026-08-16T16:56:47.794515Z","submitted_at":"2022-05-30T02:51:36Z","title":"Excess Risk of Two-Layer ReLU Neural Networks in Teacher-Student Settings and its Superiority to Kernel Methods","version":2},"cited_work":{"arxiv_id":"2205.14818","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.14818","snapshot_observed_at":"2026-08-06T14:17:38.096395Z","title":"Excess Risk of Two-Layer ReLU Neural Networks in Teacher-Student Settings and its Superiority to Kernel Methods","venue":"stat.ML","work_id":"f493d393-ea97-44cb-bbbd-ab772939f1bf","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.009819Z"},"links":{"cited_paper":"/paper/2205.14818","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:a19ed034adc9ac271cda8054dc09635a1d558a183107a2fb246695d8fb7c3f7e","observation_id":"215ebcff-3b9b-41cd-af18-ba2f9a88e71a","resolution":{"observed_at":"2026-08-06T14:17:38.139347Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.10337","last_updated":"2020-06-01T17:25:52Z","snapshot_observed_at":"2026-08-14T16:27:00.215678Z","submitted_at":"2019-05-24T17:02:51Z","title":"What Can ResNet Learn Efficiently, Going Beyond Kernels?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.10337","snapshot_observed_at":"2026-08-06T14:17:37.014664Z","title":"What can resnet learn efficiently, going beyond kernels? 0 (arXiv:1905.10337), June 2020","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.014664Z"},"links":{"cited_paper":"/paper/1905.10337","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:21ac776b23262e934dfdb59c95ddbfbc12e79c7cc7f6038e48fa923f6e56a1b3","observation_id":"26f6ad5b-063e-44cc-817b-25645fb36788","resolution":{"observed_at":"2026-08-06T14:17:37.014664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.04413","last_updated":"2023-07-07T06:12:32Z","snapshot_observed_at":"2026-08-14T04:20:38.703555Z","submitted_at":"2020-01-13T17:28:29Z","title":"Backward Feature Correction: How Deep Learning Performs Deep (Hierarchical) Learning","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.04413","snapshot_observed_at":"2026-08-06T14:17:37.019292Z","title":"Backward feature correction: How deep learning performs deep (hierarchical) learning","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.019292Z"},"links":{"cited_paper":"/paper/2001.04413","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:40ad3a1f1838ad60aaeab2dae18c54ecab328cf35050a117027022f8e0bd8394","observation_id":"c29254af-3f9c-49ba-90a1-6d36a13844e5","resolution":{"observed_at":"2026-08-06T14:17:37.019292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1601.03764","last_updated":"2018-12-07T17:30:03Z","snapshot_observed_at":"2026-08-14T22:14:34.679442Z","submitted_at":"2016-01-14T22:02:18Z","title":"Linear Algebraic Structure of Word Senses, with Applications to Polysemy","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1601.03764","snapshot_observed_at":"2026-08-06T14:17:37.023904Z","title":"Linear Algebraic Structure of Word Senses , with Applications to Polysemy , December 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.023904Z"},"links":{"cited_paper":"/paper/1601.03764","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:d792d0a52f775ce92253c594624587afd8dc28c75029a691ec0cd8ff87207b4a","observation_id":"b09f77b0-957d-4be2-b64f-50b9328f696a","resolution":{"observed_at":"2026-08-06T14:17:37.023904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.05394","last_updated":"2017-07-01T14:26:51Z","snapshot_observed_at":"2026-08-17T21:49:00.461467Z","submitted_at":"2017-06-16T18:11:09Z","title":"A Closer Look at Memorization in Deep Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.05394","snapshot_observed_at":"2026-08-06T14:17:37.028075Z","title":"Kanwal, Tegan Maharaj, Asja Fischer, Aaron Courville, Yoshua Bengio, and Simon Lacoste-Julien","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.028075Z"},"links":{"cited_paper":"/paper/1706.05394","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:d20296dd4531701df833e8181ae9a2824dc86df92a6625a6aa3c2c2dd2ddbd6d","observation_id":"d76cbb8c-d03f-4d7c-9570-21605a0d12f2","resolution":{"observed_at":"2026-08-06T14:17:37.028075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.04642","last_updated":"2025-03-02T18:16:48Z","snapshot_observed_at":"2026-08-16T13:12:06.645270Z","submitted_at":"2024-10-06T22:30:14Z","title":"The Optimization Landscape of SGD Across the Feature Learning Strength","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04642","snapshot_observed_at":"2026-08-06T14:17:37.032569Z","title":"Simon, and Cengiz Pehlevan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.032569Z"},"links":{"cited_paper":"/paper/2410.04642","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:82489094460391234f21e6d1a9b31f2aa752919c912b906b42c3a990be90c26e","observation_id":"cf62f218-02a6-4072-814a-1f2f5f07d01f","resolution":{"observed_at":"2026-08-06T14:17:37.032569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:41.427721Z","title":"Frequency bias in neural networks for input of non-uniform density","venue":null,"work_id":"edf7ae65-0dfd-406b-a6d7-fb969569d87a","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.037221Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:4f74fc2b3b60c0258e233ef5ac5a166f0ccba35b82920b9e654264e3d671980e","observation_id":"03b9d682-2d1d-483a-9bdf-29eac526326d","resolution":{"observed_at":"2026-08-06T14:17:41.517191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:41.260098Z","title":"Spectrum dependent learning curves in kernel regression and wide neural networks","venue":null,"work_id":"0c00c8b2-35ec-43cb-be37-ae66daf19de7","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.040916Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:52f124ad69c2cdb542c6340a147aab84217e63fa0ded6e072c337c35749e9098","observation_id":"8f97ebed-f8ec-4f8c-af1d-9596aadbbe02","resolution":{"observed_at":"2026-08-06T14:17:41.351332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17858","last_updated":"2025-04-04T13:47:57Z","snapshot_observed_at":"2026-08-16T13:15:07.899366Z","submitted_at":"2024-09-26T14:05:32Z","title":"How Feature Learning Can Improve Neural Scaling Laws","version":2},"cited_work":{"arxiv_id":"2409.17858","doi":"10.48550/arxiv.2409.17858","metadata_source":"pith","pith_arxiv_id":"2409.17858","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"How Feature Learning Can Improve Neural Scaling Laws","venue":"stat.ML","work_id":"ab46f089-61b7-495d-810d-6f85417722b2","year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.043915Z"},"links":{"cited_paper":"/paper/2409.17858","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:643bc832e4f36937c76a15575b82fd2ee3b97fd61138c0cec1fa4d3c123bb5fd","observation_id":"f8d0242f-3bae-4a32-992b-7f413e25c6c9","resolution":{"observed_at":"2026-08-06T14:17:37.808219Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.047386Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.047386Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:75c519aa1d0675c7dbe56944bfe571fe236fc485363290dab4fcb68e42631c58","observation_id":"92e7617d-8833-443b-a5bb-1d5b3e0d1490","resolution":{"observed_at":"2026-08-06T14:17:37.047386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.051053Z","title":"A kernel analysis of feature learning in deep neural networks","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.051053Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:054e0c1e38dec86c58ff80ddef109b449edcdaa14716ab0f392dc0775d56cd40","observation_id":"eb472d65-e66c-44c2-b7e7-12bb13d1d036","resolution":{"observed_at":"2026-08-06T14:17:37.051053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.054608Z","title":"Spectral bias and task-model alignment explain generalization in kernel regression and infinitely wide neural networks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.054608Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:b98a9a3ef560b4664e011926a7d54f0fce1dca9b3819692c364556ccd4f06a71","observation_id":"f814fb89-ea8b-4626-b190-7fb15ee5aedf","resolution":{"observed_at":"2026-08-06T14:17:37.054608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1812.07956","last_updated":"2020-01-07T16:11:56Z","snapshot_observed_at":"2026-08-14T17:41:13.899637Z","submitted_at":"2018-12-19T14:11:20Z","title":"On Lazy Training in Differentiable Programming","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1812.07956","snapshot_observed_at":"2026-08-06T14:17:37.058057Z","title":"On Lazy Training in Differentiable Programming , January 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.058057Z"},"links":{"cited_paper":"/paper/1812.07956","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:03f74204698f191f5bf8d9cc10439469a7c984163c1a5ee3f67535c17e010585","observation_id":"bde4bf20-aa6d-462c-a3f1-84dde0c4d658","resolution":{"observed_at":"2026-08-06T14:17:37.058057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.05301","last_updated":"2020-11-26T12:02:04Z","snapshot_observed_at":"2026-08-14T16:16:36.800711Z","submitted_at":"2019-06-12T18:00:10Z","title":"Learning Curves for Deep Neural Networks: A Gaussian Field Theory Perspective","version":4},"cited_work":{"arxiv_id":"1906.05301","doi":null,"metadata_source":"pith","pith_arxiv_id":"1906.05301","snapshot_observed_at":"2026-08-06T14:17:37.960429Z","title":"Learning Curves for Deep Neural Networks: A Gaussian Field Theory Perspective","venue":"cs.LG","work_id":"99cc7005-e510-432d-954b-aa9f01d80be4","year":2019},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.061657Z"},"links":{"cited_paper":"/paper/1906.05301","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:ac4d6e026ecba10261fc1b8332293da85e4c8262b85ec7ce49965ad3dfd8bfec","observation_id":"dc8dba36-0106-4584-935e-7b99202f6054","resolution":{"observed_at":"2026-08-06T14:17:37.969198Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.15144","last_updated":"2022-06-30T09:24:02Z","snapshot_observed_at":"2026-08-16T16:49:18.035166Z","submitted_at":"2022-06-30T09:24:02Z","title":"Neural Networks can Learn Representations with Gradient Descent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.15144","snapshot_observed_at":"2026-08-06T14:17:37.064891Z","title":"Lee, and Mahdi Soltanolkotabi","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.064891Z"},"links":{"cited_paper":"/paper/2206.15144","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:e4382b13bc2805b9d49fbcb6f3e7aa0ef29061db8ee96c6fe856c58030f5a91b","observation_id":"a8fea7c1-cb6b-4a1d-a376-59941a79baec","resolution":{"observed_at":"2026-08-06T14:17:37.064891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:41.010738Z","title":"Learning parities with neural networks","venue":null,"work_id":"feeca3a0-64dd-43c2-a7f1-42f9c773a0b8","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.068469Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:951b11479d1d0f7336e8878574d9e8b0fffa38bd8e524e1867a2c64a3ae55716","observation_id":"531c56ec-4101-422c-91e9-5226da84e9bf","resolution":{"observed_at":"2026-08-06T14:17:41.140043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14623","last_updated":"2025-03-04T11:18:33Z","snapshot_observed_at":"2026-08-16T13:16:26.554379Z","submitted_at":"2024-09-22T23:19:04Z","title":"From Lazy to Rich: Exact Learning Dynamics in Deep Linear Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14623","snapshot_observed_at":"2026-08-06T14:17:37.071621Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.071621Z"},"links":{"cited_paper":"/paper/2409.14623","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:45743bd44d8438d968a9c5fe5fb48b8028af1c0eacc51a05e54598843561903d","observation_id":"912eaccd-4bb1-4b41-b64f-11e825bebde8","resolution":{"observed_at":"2026-08-06T14:17:37.071621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:40.845534Z","title":"How rotational invariance of common kernels prevents generalization in high dimensions","venue":null,"work_id":"27e7b8a7-7f73-4fb9-ab99-611676d147c5","year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.075494Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:dc3e6b04b6d4b6fdffe1a6eda5936f6e291d262939a6087920994d2353e1be34","observation_id":"d53e79a8-c90d-4458-accc-b28d341301ff","resolution":{"observed_at":"2026-08-06T14:17:40.909044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.10652","last_updated":"2022-09-21T20:49:26Z","snapshot_observed_at":"2026-08-16T21:36:28.067615Z","submitted_at":"2022-09-21T20:49:26Z","title":"Toy Models of Superposition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.10652","snapshot_observed_at":"2026-08-06T14:17:37.078613Z","title":"Toy Models of Superposition , September 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.078613Z"},"links":{"cited_paper":"/paper/2209.10652","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:e1ea28757005acf9036dbd17c5a086506bb8cd1c8e00a30cd7b904ac95e312f6","observation_id":"7646d6e3-5639-4180-beb3-c2e60f38db05","resolution":{"observed_at":"2026-08-06T14:17:37.078613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05872","last_updated":"2024-07-16T17:40:09Z","snapshot_observed_at":"2026-08-16T13:36:11.410804Z","submitted_at":"2024-07-08T12:32:51Z","title":"Scaling Exponents Across Parameterizations and Optimizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05872","snapshot_observed_at":"2026-08-06T14:17:37.081920Z","title":"Alemi, Roman Novak, Peter J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.081920Z"},"links":{"cited_paper":"/paper/2407.05872","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:8f6e60d2b5620b0cc158e40159bbbd7b559125e91744ed2b6d28d3560e4a480f","observation_id":"1af5ec08-b7dd-43b3-acae-d35d9369cca7","resolution":{"observed_at":"2026-08-06T14:17:37.081920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10761","last_updated":"2024-05-17T13:17:48Z","snapshot_observed_at":"2026-08-16T13:51:52.400944Z","submitted_at":"2024-05-17T13:17:48Z","title":"Critical feature learning in deep neural networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.10761","snapshot_observed_at":"2026-08-06T14:17:37.085319Z","title":"Critical feature learning in deep neural networks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.085319Z"},"links":{"cited_paper":"/paper/2405.10761","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:eedd2e23f2e9b3e0afaa49dd7607bffc053cf878d28a85875d25015556fc73e6","observation_id":"13c27667-5119-417a-a95d-b883c7d848fa","resolution":{"observed_at":"2026-08-06T14:17:37.085319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:40.646367Z","title":"Random feature ampliﬁcation: Feature learning and generalization in neural networks","venue":null,"work_id":"f7a3d724-95c0-4206-b5e2-16ebae074de5","year":null},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.088599Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:d03ebb7a8fb5c1dab34f0b738d41844fbcddf65ed74354672c8dc675b9b70a59","observation_id":"246f93e5-70a6-4b46-a474-5d70265b7ea3","resolution":{"observed_at":"2026-08-06T14:17:40.741860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.09028","last_updated":"2022-09-28T00:10:02Z","snapshot_observed_at":"2026-08-16T17:20:24.255004Z","submitted_at":"2022-02-18T05:21:28Z","title":"On the Implicit Bias Towards Minimal Depth of Deep Neural Networks","version":9},"cited_work":{"arxiv_id":"2202.09028","doi":"10.48550/arxiv.2202.09028","metadata_source":"pith","pith_arxiv_id":"2202.09028","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"On the Implicit Bias Towards Minimal Depth of Deep Neural Networks","venue":"cs.LG","work_id":"b05c5113-7e2c-4ff1-b42b-07e53bd19545","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.091885Z"},"links":{"cited_paper":"/paper/2202.09028","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:e6099c72c7a6c248ba82712a49c50540d5f314bba95da11437e1bca13c557495","observation_id":"a3a0aed8-e6bf-4214-9fb7-f7ddc5906804","resolution":{"observed_at":"2026-08-06T14:17:37.762133Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.09255","last_updated":"2022-03-17T11:23:18Z","snapshot_observed_at":"2026-08-16T17:13:49.586802Z","submitted_at":"2022-03-17T11:23:18Z","title":"On the Spectral Bias of Convolutional Neural Tangent and Gaussian Process Kernels","version":1},"cited_work":{"arxiv_id":"2203.09255","doi":"10.48550/arxiv.2203.09255","metadata_source":"pith","pith_arxiv_id":"2203.09255","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"On the Spectral Bias of Convolutional Neural Tangent and Gaussian Process Kernels","venue":"cs.LG","work_id":"be2c6f4a-7440-493c-87f5-16df5c98deb2","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.096130Z"},"links":{"cited_paper":"/paper/2203.09255","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:b3060e4d99634118b13d00df8576f32ecc4d1cf0309cc9cc08ac1765f64ec52c","observation_id":"ba78a3c7-f7eb-4111-bedb-592dac8181e2","resolution":{"observed_at":"2026-08-06T14:17:37.751175Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.14531","last_updated":"2024-03-20T07:49:41Z","snapshot_observed_at":"2026-08-16T15:12:59.110033Z","submitted_at":"2023-07-26T22:39:47Z","title":"Controlling the Inductive Bias of Wide Neural Networks by Modifying the Kernel's Spectrum","version":2},"cited_work":{"arxiv_id":"2307.14531","doi":"10.48550/arxiv.2307.14531","metadata_source":"pith","pith_arxiv_id":"2307.14531","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Controlling the Inductive Bias of Wide Neural Networks by Modifying the Kernel's Spectrum","venue":"cs.LG","work_id":"ec6c679a-87b2-4416-a6c2-4bb109e12bbd","year":2023},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.099870Z"},"links":{"cited_paper":"/paper/2307.14531","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:314fc2c62ca2d8ce74e08bf3aa15656633f3dcf07100a3606a5dc589d5c7107b","observation_id":"a3a44a7b-845b-4e4e-b7db-0c772e632703","resolution":{"observed_at":"2026-08-06T14:17:37.738925Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1906.08034","last_updated":"2020-10-04T19:22:13Z","snapshot_observed_at":"2026-08-14T17:59:29.314748Z","submitted_at":"2019-06-19T12:08:13Z","title":"Disentangling feature and lazy training in deep neural networks","version":4},"cited_work":{"arxiv_id":"1906.08034","doi":null,"metadata_source":"pith","pith_arxiv_id":"1906.08034","snapshot_observed_at":"2026-08-06T14:17:37.916025Z","title":"Disentangling feature and lazy training in deep neural networks","venue":"cs.LG","work_id":"bb34c329-a758-4804-8f2c-52296ba60bcf","year":2019},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.103459Z"},"links":{"cited_paper":"/paper/1906.08034","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:aa5e1ce5370873bbeb58ec6093e127fa34c912a147df6e875612f4a596a15212","observation_id":"b865dea3-69dd-41b7-a0ac-7f377d5d5f08","resolution":{"observed_at":"2026-08-06T14:17:37.928319Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:40.432263Z","title":"When do neural networks outperform kernel methods? In Advances in Neural Information Processing Systems, volume 33, page 14820–14830","venue":null,"work_id":"ed886682-eede-44b3-bebf-d92b4008bab7","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.107023Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:c2cf3c9c53e62fcc862e5e5f7518c621cde46ff0f058d25d493a697f2f2d8f8a","observation_id":"56288377-77f8-468f-8489-79d1735c94db","resolution":{"observed_at":"2026-08-06T14:17:40.535237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.08384","last_updated":"2022-02-17T00:20:12Z","snapshot_observed_at":"2026-08-16T17:20:42.756230Z","submitted_at":"2022-02-17T00:20:12Z","title":"Limitations of Neural Collapse for Understanding Generalization in Deep Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.08384","snapshot_observed_at":"2026-08-06T14:17:37.110654Z","title":"Limitations of neural collapse for understanding generalization in deep learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.110654Z"},"links":{"cited_paper":"/paper/2202.08384","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:2ad05a2b0017df32d210ed2126dd96ea43381c78d6986e1c6915dfdd520b32df","observation_id":"28ae4097-2800-4b92-ba9b-1011070948aa","resolution":{"observed_at":"2026-08-06T14:17:37.110654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05451","last_updated":"2024-08-10T06:11:48Z","snapshot_observed_at":"2026-08-16T13:27:19.281750Z","submitted_at":"2024-08-10T06:11:48Z","title":"Mathematical Models of Computation in Superposition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05451","snapshot_observed_at":"2026-08-06T14:17:37.115036Z","title":"Mathematical Models of Computation in Superposition , August 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.115036Z"},"links":{"cited_paper":"/paper/2408.05451","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:5acd370e9133a596bbbca0f7caf3b4e1502f82c472b80d54bd75d6d56085cddf","observation_id":"9b410276-89de-456a-a04c-c31374e792b5","resolution":{"observed_at":"2026-08-06T14:17:37.115036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:40.294350Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":"05948904-9035-4730-8e25-7c994133200f","year":2018},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.119424Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:e8147c0beba5bf1bfe36c248927e345b839be45067fed9f339e6bbd3d3db3e99","observation_id":"3764a69c-3cfa-4d83-9e73-06e62b01ad77","resolution":{"observed_at":"2026-08-06T14:17:40.347764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.122585Z","title":"Highly accurate protein structure prediction with alphafold","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.122585Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:292df905c106019bf2ce181067884d64c359843bbc92a8096b5434b28705102e","observation_id":"7bbf47d7-0c8f-456b-86ea-79b753d28159","resolution":{"observed_at":"2026-08-06T14:17:37.122585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19719","last_updated":"2024-10-07T18:14:21Z","snapshot_observed_at":"2026-08-16T13:56:31.860147Z","submitted_at":"2024-04-30T17:11:12Z","title":"The lazy (NTK) and rich ($\\mu$P) regimes: a gentle tutorial","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.19719","snapshot_observed_at":"2026-08-06T14:17:37.126047Z","title":"The lazy ( NTK ) and rich ( P ) regimes: a gentle tutorial, October 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.126047Z"},"links":{"cited_paper":"/paper/2404.19719","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:5f587ac72e0f7084922e4b043e4d7d7b81db8818444038e833270d05df574a88","observation_id":"1bc9c924-79bd-41dd-ba7f-274bc8dc420c","resolution":{"observed_at":"2026-08-06T14:17:37.126047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.00414","last_updated":"2019-07-19T14:59:45Z","snapshot_observed_at":"2026-08-18T09:29:11.457911Z","submitted_at":"2019-05-01T17:57:26Z","title":"Similarity of Neural Network Representations Revisited","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.00414","snapshot_observed_at":"2026-08-06T14:17:37.129479Z","title":"Similarity of neural network representations revisited","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.129479Z"},"links":{"cited_paper":"/paper/1905.00414","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:f8af18ee62befc24ff3c162fe115cee9e07f0e79ee5ff6d060284cd414e35dfe","observation_id":"146a10a5-e4c7-4103-ae94-635abcc860ed","resolution":{"observed_at":"2026-08-06T14:17:37.129479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04041","last_updated":"2023-04-11T06:11:14Z","snapshot_observed_at":"2026-08-16T16:54:23.685015Z","submitted_at":"2022-06-08T17:55:28Z","title":"Neural Collapse: A Review on Modelling Principles and Generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04041","snapshot_observed_at":"2026-08-06T14:17:37.133481Z","title":"Neural collapse: A review on modelling principles and generalization","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.133481Z"},"links":{"cited_paper":"/paper/2206.04041","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:0384f402c1439582485b6607ecfac09dbe7e759ad3be87d8d506c10818faf1e0","observation_id":"f0e1369d-40ec-478c-82bf-94d7d5869042","resolution":{"observed_at":"2026-08-06T14:17:37.133481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.136671Z","title":"Deep learning","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.136671Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:4894f9761718404f8f825d825a0dd9c1563fd7fc3621ba2eedca6d9cf05e5031","observation_id":"0ef157b4-f3a6-479c-a172-f49b83b1d175","resolution":{"observed_at":"2026-08-06T14:17:37.136671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.00165","last_updated":"2018-03-03T00:45:00Z","snapshot_observed_at":"2026-08-14T20:18:09.717440Z","submitted_at":"2017-11-01T02:13:25Z","title":"Deep Neural Networks as Gaussian Processes","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.00165","snapshot_observed_at":"2026-08-06T14:17:37.139654Z","title":"Schoenholz, Jeffrey Pennington, and Jascha Sohl-Dickstein","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.139654Z"},"links":{"cited_paper":"/paper/1711.00165","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:404d8c4f78df3d73374afc76787eb90836c3fce46d498d3691e038d90b704a96","observation_id":"5e24baa8-8072-47d9-8375-c7361d51d3e1","resolution":{"observed_at":"2026-08-06T14:17:37.139654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.04596","last_updated":"2020-07-09T07:09:28Z","snapshot_observed_at":"2026-08-17T13:12:00.620113Z","submitted_at":"2020-07-09T07:09:28Z","title":"Learning Over-Parametrized Two-Layer ReLU Neural Networks beyond NTK","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.04596","snapshot_observed_at":"2026-08-06T14:17:37.143199Z","title":null,"venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.143199Z"},"links":{"cited_paper":"/paper/2007.04596","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:eecff5078fe7b6f059141f7dd6e71463fa36e7f8f281e9d47f429c10ce3aae05","observation_id":"b5ec147e-af05-42d5-816c-2628dc1b0be0","resolution":{"observed_at":"2026-08-06T14:17:37.143199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18817","last_updated":"2024-04-02T05:43:18Z","snapshot_observed_at":"2026-08-16T14:38:42.411411Z","submitted_at":"2023-11-30T18:55:38Z","title":"Dichotomy of Early and Late Phase Implicit Biases Can Provably Induce Grokking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18817","snapshot_observed_at":"2026-08-06T14:17:37.147161Z","title":"Du, Jason D","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.147161Z"},"links":{"cited_paper":"/paper/2311.18817","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:771c5a199669dd34a59956fa68c960fa7e498e2899f66cef8a999b448ce1032f","observation_id":"9ab219f1-b177-4de1-918e-9cb3a955493a","resolution":{"observed_at":"2026-08-06T14:17:37.147161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.01210","last_updated":"2021-03-01T18:54:13Z","snapshot_observed_at":"2026-08-16T18:42:06.490771Z","submitted_at":"2021-03-01T18:54:13Z","title":"Quantifying the Benefit of Using Differentiable Learning over Tangent Kernels","version":1},"cited_work":{"arxiv_id":"2103.01210","doi":"10.48550/arxiv.2103.01210","metadata_source":"pith","pith_arxiv_id":"2103.01210","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Quantifying the Benefit of Using Differentiable Learning over Tangent Kernels","venue":"cs.LG","work_id":"a75f69a8-5333-4b9e-9228-5bb4fd8bbf00","year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.150394Z"},"links":{"cited_paper":"/paper/2103.01210","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:27506eb84c000334bd8cd359430651233bff81e1c99507d909f8c0fdd733b009","observation_id":"e9051f66-5ab2-403d-87e1-9678921a5d5e","resolution":{"observed_at":"2026-08-06T14:17:37.681522Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:40.128743Z","title":"Implicit bias in deep linear classification: Initialization scale vs training accuracy","venue":null,"work_id":"ddae6059-c378-4f52-bf9b-29b0412e98ec","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.154114Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:c0a9d7da2310251c97d2186debcb172a9e24ac696fc814bc7a917676efe3832e","observation_id":"8f03c7ec-f29c-4258-b148-aa80e0eb1e03","resolution":{"observed_at":"2026-08-06T14:17:40.180210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:39.883430Z","title":"Neural networks efficiently learn low-dimensional representations with sgd","venue":null,"work_id":"f07a374e-6de7-4c36-8182-f40e26403450","year":2023},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.157080Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:405a2f6cda5441db57d084e74e1550eb020310e8c29c88be3adbfa8ff6728053","observation_id":"26c5b5d2-13d6-4a2b-98c7-8d718e59c47f","resolution":{"observed_at":"2026-08-06T14:17:40.024255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07254","last_updated":"2025-03-27T03:40:20Z","snapshot_observed_at":"2026-08-16T13:26:36.297408Z","submitted_at":"2024-08-14T02:13:35Z","title":"Learning Multi-Index Models with Neural Networks via Mean-Field Langevin Dynamics","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07254","snapshot_observed_at":"2026-08-06T14:17:37.159912Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.159912Z"},"links":{"cited_paper":"/paper/2408.07254","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:76020a1c87838caeb51a802b61df7090ed20d6816564e8955a275e6fb50013cd","observation_id":"76038ca5-b51d-486f-9a90-54834a59fc4f","resolution":{"observed_at":"2026-08-06T14:17:37.159912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2410.04264","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Visualising feature learning in deep neural networks by diagonalizing the forward feature map","venue":"arXiv (Cornell University)","work_id":"eafbdda9-ccf0-413f-824c-805373561479","year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.163460Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:708910f8b8437b9c307fa648ffbc983cc0910cfc272d605d3e52cdac84150254","observation_id":"e080e7b4-c1f3-449a-a7e0-3a6276f849cf","resolution":{"observed_at":"2026-08-06T14:17:37.662418Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:39.682188Z","title":"A self consistent theory of gaussian processes captures feature learning effects in finite cnns","venue":null,"work_id":"b5525b8e-ec66-43bf-9981-548db238c22d","year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.166868Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:7c47d6a0cad48c443f3933a40875b3d20fee3ac7e629d137cf1a17380bc159eb","observation_id":"0ba14e3f-0156-4630-852f-482165cf09dc","resolution":{"observed_at":"2026-08-06T14:17:39.781774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:39.505500Z","title":"Alemi, Jascha Sohl-Dickstein, and Samuel S","venue":null,"work_id":"f8193d1f-ef76-4f95-a7ed-0b176a88226f","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.170078Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:beb884d104f1c95c6ca8f4d66d09f68c9611a90bacc1db509497d35050766ef2","observation_id":"a725812b-ecf4-4c62-a210-2b2029872e46","resolution":{"observed_at":"2026-08-06T14:17:39.569805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:39.242416Z","title":"Schoenholz","venue":null,"work_id":"a3a5855f-d60c-4a65-be06-31ea4fa859bf","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.173194Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:375834b1283d7e4a8891c35d7f239d7b5e69a5532a584165494ae9f000ddef07","observation_id":"eb40b16e-1aec-46e2-a36b-2899ec97533c","resolution":{"observed_at":"2026-08-06T14:17:39.402320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.176258Z","title":"Feature visualization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.176258Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:d45c34e30bcabac77eaa08eccf439986c96f90b01784b97e90fdca0d9debd702","observation_id":"2914d680-3b47-419e-9358-13fe943e03a8","resolution":{"observed_at":"2026-08-06T14:17:37.176258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.06770","last_updated":"2021-10-13T07:25:06Z","snapshot_observed_at":"2026-08-16T19:05:08.999556Z","submitted_at":"2021-06-12T13:05:11Z","title":"What can linearized neural networks actually say about generalization?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.06770","snapshot_observed_at":"2026-08-06T14:17:37.179898Z","title":"What can linearized neural networks actually say about generalization? 0 (arXiv:2106.06770), October 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.179898Z"},"links":{"cited_paper":"/paper/2106.06770","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:7ea58a360bb539af693084b8982c7bf11ea3431fa5d8f5d3c4583c5b399c3aca","observation_id":"385df943-c7d1-40ea-9711-e466110557f5","resolution":{"observed_at":"2026-08-06T14:17:37.179898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2008.08186","last_updated":"2020-08-21T16:15:50Z","snapshot_observed_at":"2026-08-03T20:42:15.166440Z","submitted_at":"2020-08-18T23:12:54Z","title":"Prevalence of Neural Collapse during the terminal phase of deep learning training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2008.08186","snapshot_observed_at":"2026-08-06T14:17:37.183370Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.183370Z"},"links":{"cited_paper":"/paper/2008.08186","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:c61c26a12ba725636bc2eeba467da2b4529703744fb63310b14d1de87d83e926","observation_id":"1c767e82-b8fc-40e7-bc92-e3eed480adf3","resolution":{"observed_at":"2026-08-06T14:17:37.183370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.12314","last_updated":"2022-10-12T14:51:26Z","snapshot_observed_at":"2026-08-16T16:50:36.171727Z","submitted_at":"2022-06-24T14:26:33Z","title":"Learning sparse features can lead to overfitting in neural networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.12314","snapshot_observed_at":"2026-08-06T14:17:37.186905Z","title":"Learning sparse features can lead to overfitting in neural networks, October 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.186905Z"},"links":{"cited_paper":"/paper/2206.12314","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:c012e58fc9ab0691091f302804160bad3d23e637669b900590067a53f1a0895e","observation_id":"fdf39048-1f06-4a38-872a-1590a0dc01e4","resolution":{"observed_at":"2026-08-06T14:17:37.186905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:39.070680Z","title":"On the spectral bias of neural networks","venue":null,"work_id":"3abe21f7-2155-4642-9aa2-bf7dad48ea77","year":null},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.190284Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:b94e78a02c9cb479d5961514a92dabc701aaa2aac10b9497952927ba4cb5549c","observation_id":"b68433be-c47f-44d5-b1d0-f390fdacc10a","resolution":{"observed_at":"2026-08-06T14:17:39.172776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.11742","last_updated":"2021-06-10T16:24:03Z","snapshot_observed_at":"2026-08-16T18:43:40.456507Z","submitted_at":"2021-02-23T15:10:15Z","title":"Classifying high-dimensional Gaussian mixtures: Where kernel methods fail and neural networks succeed","version":2},"cited_work":{"arxiv_id":"2102.11742","doi":"10.48550/arxiv.2102.11742","metadata_source":"pith","pith_arxiv_id":"2102.11742","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Classifying high-dimensional Gaussian mixtures: Where kernel methods fail and neural networks succeed","venue":"cs.LG","work_id":"d9a5c2af-44e5-40f3-ba95-a7e077e3043a","year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.193874Z"},"links":{"cited_paper":"/paper/2102.11742","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:96d9d7d7255d1108badd421f7c446e81d78c404d99bac6e5de9de731556e4d6a","observation_id":"6940197f-a7b6-4a1d-82a7-3a7610951614","resolution":{"observed_at":"2026-08-06T14:17:37.581135Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:38.928285Z","title":"Analyzing finite neural networks: Can we trust neural tangent kernel theory?","venue":null,"work_id":"2982cc1d-f114-448d-92d6-ff7439b4b6e5","year":null},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.197855Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:22182d813f0cf894a5f0848420ad64df80ebdf3bc714727163542d9cea1c3a78","observation_id":"be64cf14-2496-4447-953a-dffb0db11d89","resolution":{"observed_at":"2026-08-06T14:17:38.995024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.15383","last_updated":"2022-09-22T21:42:18Z","snapshot_observed_at":"2026-08-16T17:30:57.743783Z","submitted_at":"2021-12-31T10:49:55Z","title":"Separation of Scales and a Thermodynamic Description of Feature Learning in Some CNNs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.15383","snapshot_observed_at":"2026-08-06T14:17:37.201402Z","title":"Separation of scales and a thermodynamic description of feature learning in some cnns","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.201402Z"},"links":{"cited_paper":"/paper/2112.15383","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:3a726e6058bf343002de9956fb0a778f183c6b1bdb30a07211e20c1ef0e3f635","observation_id":"a5e63c65-cde8-4c1b-b84c-98a24efe1a3b","resolution":{"observed_at":"2026-08-06T14:17:37.201402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:37.204996Z","title":"Separation of scales and a thermodynamic description of feature learning in some cnns","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.204996Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:93677077157fe105ade214533cb0032470782a2ca37c8af6f658683afaee3590","observation_id":"af256f2f-4684-4dc6-9b86-269b8e958115","resolution":{"observed_at":"2026-08-06T14:17:37.204996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.01717","last_updated":"2022-06-03T17:49:38Z","snapshot_observed_at":"2026-08-16T16:55:26.935338Z","submitted_at":"2022-06-03T17:49:38Z","title":"A Theoretical Analysis on Feature Learning in Neural Networks: Emergence from Inputs and Advantage over Fixed Features","version":1},"cited_work":{"arxiv_id":"2206.01717","doi":"10.48550/arxiv.2206.01717","metadata_source":"pith","pith_arxiv_id":"2206.01717","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"A Theoretical Analysis on Feature Learning in Neural Networks: Emergence from Inputs and Advantage over Fixed Features","venue":"cs.LG","work_id":"92330ae1-4d9e-41d3-9865-f54e47f790c5","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.208534Z"},"links":{"cited_paper":"/paper/2206.01717","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:629685d868996002e712ce0e8ee0d319c2359f3ec25ebdfd0a563b6037bb8206","observation_id":"fb5dc8c5-2d2a-4888-aad6-f1915755851c","resolution":{"observed_at":"2026-08-06T14:17:37.553585Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1905.10843","last_updated":"2020-08-18T13:27:19Z","snapshot_observed_at":"2026-08-17T04:54:09.008289Z","submitted_at":"2019-05-26T17:29:11Z","title":"Asymptotic learning curves of kernel methods: empirical data v.s. Teacher-Student paradigm","version":8},"cited_work":{"arxiv_id":"1905.10843","doi":null,"metadata_source":"pith","pith_arxiv_id":"1905.10843","snapshot_observed_at":"2026-08-06T14:17:37.860838Z","title":"Asymptotic learning curves of kernel methods: empirical data v.s. Teacher-Student paradigm","venue":"stat.ML","work_id":"1db64e28-e7b0-496e-9467-0b69c3922b08","year":2019},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.212050Z"},"links":{"cited_paper":"/paper/1905.10843","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:85b66fbc7d5faca88cf0fd92893ad443614f8d00a4109b1be8ba219c3942d87d","observation_id":"5e159d8c-0aad-42b2-b8a6-f655c4ae8a82","resolution":{"observed_at":"2026-08-06T14:17:37.869152Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14922","last_updated":"2024-10-10T14:27:40Z","snapshot_observed_at":"2026-08-17T12:43:28.592453Z","submitted_at":"2023-12-22T18:55:25Z","title":"Learning from higher-order statistics, efficiently: hypothesis tests, random features, and neural networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14922","snapshot_observed_at":"2026-08-06T14:17:37.215595Z","title":"Learning from higher-order statistics, efficiently: hypothesis tests, random features, and neural networks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.215595Z"},"links":{"cited_paper":"/paper/2312.14922","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:9e3dfe98c0df52ed14125cb78380de6b25040c2aa739387abc04b3dada442d70","observation_id":"a941fc44-d26b-4c11-9991-fa5cb3f08abe","resolution":{"observed_at":"2026-08-06T14:17:37.215595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:38.752837Z","title":"Feature selection and low test error in shallow low-rotation relu networks","venue":null,"work_id":"7b342f14-27d4-4daf-8be9-bfcf496703a6","year":null},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.219237Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:6ca7f1071a187379cde0d8d0d7cbfabac453eb7ea9cdee9a141c853c6cc097d9","observation_id":"654be64e-9764-4cf3-b74b-43ed9d71f90d","resolution":{"observed_at":"2026-08-06T14:17:38.845389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.03348","last_updated":"2022-02-16T11:25:25Z","snapshot_observed_at":"2026-08-16T17:23:03.863984Z","submitted_at":"2022-02-07T16:48:14Z","title":"Failure and success of the spectral bias prediction for Kernel Ridge Regression: the case of low-dimensional data","version":2},"cited_work":{"arxiv_id":"2202.03348","doi":"10.48550/arxiv.2202.03348","metadata_source":"pith","pith_arxiv_id":"2202.03348","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Failure and success of the spectral bias prediction for Kernel Ridge Regression: the case of low-dimensional data","venue":"cs.LG","work_id":"ee6cc1b8-5cc1-44c0-b8b9-71cdbdc29018","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.222460Z"},"links":{"cited_paper":"/paper/2202.03348","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:71b74d1860732651a8c42304bd99e6d1c557ceae98cb766ad9bb9f8a269c2ad2","observation_id":"889f29ad-b041-435d-a841-2fa2fc15675d","resolution":{"observed_at":"2026-08-06T14:17:37.541736Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2405.15480","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Fundamental computational limits of weak learnability in high-dimensional multi-index models","venue":"arXiv (Cornell University)","work_id":"44592cce-05b4-4095-98d2-3ace9a739ec6","year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.226936Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:ee1400af1d042551772e26af4a76649dccc62181ae815a412bb0f0b59e0d5305","observation_id":"0b6e347f-1e27-4575-a616-1fa15168b497","resolution":{"observed_at":"2026-08-06T14:17:37.527764Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17580","last_updated":"2024-10-29T20:52:18Z","snapshot_observed_at":"2026-08-16T13:48:54.478436Z","submitted_at":"2024-05-27T18:29:23Z","title":"Mixed Dynamics In Linear Networks: Unifying the Lazy and Active Regimes","version":2},"cited_work":{"arxiv_id":"2405.17580","doi":"10.48550/arxiv.2405.17580","metadata_source":"pith","pith_arxiv_id":"2405.17580","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Mixed Dynamics In Linear Networks: Unifying the Lazy and Active Regimes","venue":"cs.LG","work_id":"192b8a77-cb98-4d8b-a0a2-c82cfc22e6c9","year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.230492Z"},"links":{"cited_paper":"/paper/2405.17580","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:d8aea3e544f8c1bcc1593d4d9db65ce32175ead115678dedd4e9d37b197fcbe4","observation_id":"372fb401-64c8-455d-9e95-69da73a91855","resolution":{"observed_at":"2026-08-06T14:17:37.352664Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10012","last_updated":"2022-06-20T21:23:28Z","snapshot_observed_at":"2026-08-16T16:51:39.151684Z","submitted_at":"2022-06-20T21:23:28Z","title":"Limitations of the NTK for Understanding Generalization in Deep Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10012","snapshot_observed_at":"2026-08-06T14:17:37.234144Z","title":"Limitations of the ntk for understanding generalization in deep learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.234144Z"},"links":{"cited_paper":"/paper/2206.10012","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:a005f55ceaf1371d04219a752823a9b158fa5bc7eca31d11896bd2b573554a21","observation_id":"2bdd98f0-9d2a-4d3b-956b-480b6a9697d8","resolution":{"observed_at":"2026-08-06T14:17:37.234144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11356","last_updated":"2020-10-22T00:32:12Z","snapshot_observed_at":"2026-08-16T19:11:38.145121Z","submitted_at":"2020-10-22T00:32:12Z","title":"Beyond Lazy Training for Over-parameterized Tensor Decomposition","version":1},"cited_work":{"arxiv_id":"2010.11356","doi":"10.48550/arxiv.2010.11356","metadata_source":"pith","pith_arxiv_id":"2010.11356","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Beyond Lazy Training for Over-parameterized Tensor Decomposition","venue":"stat.ML","work_id":"4db40ffa-a3f5-4c98-829d-7c1106e12f64","year":2020},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.237948Z"},"links":{"cited_paper":"/paper/2010.11356","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:24565dad9eada884b1904fdb29eebdeb26d396e65e5b76542968b4c00144b710","observation_id":"13bad243-2c1c-4a16-b8fe-04409858a5f8","resolution":{"observed_at":"2026-08-06T14:17:37.329712Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:38.575988Z","title":"More than a toy: Random matrix models predict how real-world neural representations generalize","venue":null,"work_id":"50a9cbbe-1d86-4f3f-b921-df26c98d20ae","year":2022},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.241987Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:4e72322ef582b173819a26ccb62b06544f43bd201c01c6242f364dbefd46b2ee","observation_id":"5764a9b1-07fc-4849-bd0c-c33aac7729cf","resolution":{"observed_at":"2026-08-06T14:17:38.656933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:38.401973Z","title":"Regularization matters: Generalization and optimization of neural nets v.s","venue":null,"work_id":"6c857898-c2ad-462f-bada-c38849c271b1","year":2019},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.245343Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:4bb4bf43a58a0338509421acaa91ce78fa19853a391d443c504b1bf42d2b3933","observation_id":"ef8c8e6e-cb06-427a-b578-23767bcd07da","resolution":{"observed_at":"2026-08-06T14:17:38.495149Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00137","last_updated":"2024-05-28T20:59:41Z","snapshot_observed_at":"2026-08-16T14:56:22.413472Z","submitted_at":"2023-09-29T20:51:24Z","title":"On the Disconnect Between Theory and Practice of Neural Networks: Limits of the NTK Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00137","snapshot_observed_at":"2026-08-06T14:17:37.249270Z","title":"On the disconnect between theory and practice of neural networks: Limits of the ntk perspective","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.249270Z"},"links":{"cited_paper":"/paper/2310.00137","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:ce7c5aeb21b9417c64bf5e80878022d6b29a2f312caceeb231ee4fb53fcbb612","observation_id":"4f56b5ce-177c-40c5-bed5-5433383538a8","resolution":{"observed_at":"2026-08-06T14:17:37.249270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:17:38.221586Z","title":null,"venue":null,"work_id":"f9213b2f-f0e3-4f91-aa2b-1d4973a7ebfb","year":2021},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.252618Z"},"links":{"citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:f278ea44941de65be5279a0cd1bb21ef7e53213cba71249f6dbfb3fec208330a","observation_id":"4aeb6ab2-def9-47aa-970f-0b5d7a180284","resolution":{"observed_at":"2026-08-06T14:17:38.312426Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.00687","last_updated":"2022-02-27T11:55:01Z","snapshot_observed_at":"2026-08-16T21:32:46.680916Z","submitted_at":"2019-04-01T10:21:24Z","title":"On the Power and Limitations of Random Features for Understanding Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.00687","snapshot_observed_at":"2026-08-06T14:17:37.255620Z","title":"On the power and limitations of random features for understanding neural networks","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.255620Z"},"links":{"cited_paper":"/paper/1904.00687","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:da7c33a4eabae0236fdb358e551ade789ddc311d03d6b22a027759e757f17906","observation_id":"cee354a0-a4a9-4536-9824-796fe7941253","resolution":{"observed_at":"2026-08-06T14:17:37.255620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1605.07146","last_updated":"2017-06-14T06:06:48Z","snapshot_observed_at":"2026-08-16T07:54:28.918946Z","submitted_at":"2016-05-23T19:27:13Z","title":"Wide Residual Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1605.07146","snapshot_observed_at":"2026-08-06T14:17:37.258687Z","title":"Wide residual networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.258687Z"},"links":{"cited_paper":"/paper/1605.07146","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:77f1c9ab8b7847bb45cff9735d072a13f056ff6c2437df25817f456f17754d8e","observation_id":"e7b8b28c-dfa2-4da3-8511-043710d2d55e","resolution":{"observed_at":"2026-08-06T14:17:37.258687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1611.03530","last_updated":"2017-02-26T19:36:40Z","snapshot_observed_at":"2026-08-01T16:56:59.989486Z","submitted_at":"2016-11-10T22:02:36Z","title":"Understanding deep learning requires rethinking generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.03530","snapshot_observed_at":"2026-08-06T14:17:37.261961Z","title":"Understanding deep learning requires rethinking generalization","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-06T14:17:37.261961Z"},"links":{"cited_paper":"/paper/1611.03530","citing_paper":"/paper/2507.19680"},"observation_digest":"sha256:5a60e1585a1224caf17790df2a0ec3a031dee29076ee89521300e9de6494724e","observation_id":"e91c8c6b-dba0-4203-bf23-fc0aca4c58ef","resolution":{"observed_at":"2026-08-06T14:17:37.261961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.19680","last_updated":"2025-07-25T21:19:37Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-16T06:57:12.804004Z","submitted_at":"2025-07-25T21:19:37Z","title":"Feature learning is decoupled from generalization in high capacity neural networks"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":40,"verified_exact":14,"verified_fuzzy":17},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 0 inbound Pith citation observations for arXiv:2507.19680."}