{"as_of":"2026-08-07T05:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cf7298d4e6900f718bd53a26b83e737196fe9fb1af2803447241ba9335b1f08a","coverage":[{"denominator":53,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":53,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:31:23.888184Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T22:53:30.561450Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-29T22:54:00.682463Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"cited_work":{"arxiv_id":"2507.05644","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.05644","snapshot_observed_at":"2026-06-29T22:54:00.682463Z","title":"Boix-Adsera, N","venue":null,"work_id":"5b61d926-bb01-4536-adf4-04c6e8862bff","year":2026},"citing_paper":{"arxiv_id":"2606.07563","last_updated":"2026-05-25T18:32:52Z","snapshot_observed_at":"2026-07-06T23:47:10.030985Z","submitted_at":"2026-05-25T18:32:52Z","title":"Emergence via Phase Transitions: Mechanism Landscapes and Universal Convergence Across Complex Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T22:53:30.561450Z"},"links":{"cited_paper":"/paper/2507.05644","citing_paper":"/paper/2606.07563"},"observation_digest":"sha256:6e1ba35c87aced55d0657c38555c2c4a8dcff1d18850f1b1dd4bc46f0c966648","observation_id":"ff9315d3-5ca8-48f3-8eec-64cd09cffed3","resolution":{"observed_at":"2026-06-29T22:54:00.683913Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.05644/citation-record","integrity":"/paper/2507.05644/integrity","json":"/paper/2507.05644/citation-record.json","paper":"/paper/2507.05644"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:25.033265Z","title":null,"venue":null,"work_id":"84420f34-53ec-4f20-a43f-478a68f9cad6","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.105519Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:291317326e9edd999998c6b505396ee3cc7fa6bbad07e79b2940597c961cc122","observation_id":"d2a7dd51-c5bb-4e64-ba91-6e51baf603be","resolution":{"observed_at":"2026-08-06T19:31:25.037471Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:25.019473Z","title":null,"venue":null,"work_id":"dd41bd11-337e-448d-a12d-7ab11d6298d3","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.205740Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:557743d37655fc7603c8f4b891ddf2c53209719e022a2d2beb35615fd9b87540","observation_id":"8dc1e045-a237-4cdb-9d54-d6c85e0aae40","resolution":{"observed_at":"2026-08-06T19:31:25.023591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:25.005712Z","title":"Arora, N","venue":null,"work_id":"f31417e3-069e-4f59-a472-0fae38cae75b","year":2019},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.338909Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:9aa88a4d35cb32a0bbb499c8aaf9bb11982d44df52a75970ef82caf6a7d1e775","observation_id":"1c2029c5-56d5-46d8-b07b-08b1e357365f","resolution":{"observed_at":"2026-08-06T19:31:25.009845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.991963Z","title":"Arora, N","venue":null,"work_id":"8efa5d94-0487-4b6e-9e22-174aceba9658","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.437898Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:8b35c51f579154b55dd649c902c7aa342a80fb96561033a490ffd1e2fb998207","observation_id":"7dc1aa41-710e-4d04-9812-e03eea3f77be","resolution":{"observed_at":"2026-08-06T19:31:24.996016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.977939Z","title":"Arora, N","venue":null,"work_id":"76422e8b-7bf7-4c38-bc92-41f825032c50","year":2019},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.539500Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:d901c50905f0c2780d86f671c61ffd8a3c97acb2d6702f87e36c889d4a3907bc","observation_id":"aefc0213-27b4-476a-8c64-43c8cf7fb730","resolution":{"observed_at":"2026-08-06T19:31:24.982231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.964522Z","title":null,"venue":null,"work_id":"22085176-33fb-483f-a423-c3b63b8e1710","year":2021},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.640051Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:8a8333dba3dc065891da6de3b35987901d562e5dd8552c030c6cd6a7c0896d35","observation_id":"6e12cb4f-73fb-49c0-9a85-0223646c5a95","resolution":{"observed_at":"2026-08-06T19:31:24.968565Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.950424Z","title":"Barak, B","venue":null,"work_id":"e1a5d074-1b8c-4176-a0cb-58b55f14ce51","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.786168Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:6bce7513d2ef1a30ff9b84292aec40d73b225994049896b642e6f366f92368d7","observation_id":"fc3c05ed-b535-4bb1-bc4e-601881048331","resolution":{"observed_at":"2026-08-06T19:31:24.955034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03708","last_updated":"2025-05-28T19:52:55Z","snapshot_observed_at":"2026-07-06T20:31:56.011023Z","submitted_at":"2025-02-06T01:41:48Z","title":"Toward universal steering and monitoring of AI models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.03708","snapshot_observed_at":"2026-08-06T19:31:18.912515Z","title":"Beaglehole, A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:18.912515Z"},"links":{"cited_paper":"/paper/2502.03708","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:0c810600909ba71efa58b1456643dace7c5c65bdadd1830e0f757884856207cc","observation_id":"72763c8e-e5cf-4e8f-9e15-48f13c44e7d0","resolution":{"observed_at":"2026-08-06T19:31:18.912515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00570","last_updated":"2023-09-01T16:30:02Z","snapshot_observed_at":"2026-08-07T03:23:02.980747Z","submitted_at":"2023-09-01T16:30:02Z","title":"Mechanism of feature learning in convolutional neural networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00570","snapshot_observed_at":"2026-08-06T19:31:19.028549Z","title":"Beaglehole, A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.028549Z"},"links":{"cited_paper":"/paper/2309.00570","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:71bc2f7a2921bb970357793cfbe253a0e340d27757bbc033de2d5738252b1143","observation_id":"88e5ce86-8a4e-427b-ad8d-028614980bb1","resolution":{"observed_at":"2026-08-06T19:31:19.028549Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02984","last_updated":"2024-02-20T19:17:52Z","snapshot_observed_at":"2026-07-06T16:27:47.445841Z","submitted_at":"2023-10-04T17:20:34Z","title":"Scaling Laws for Associative Memories","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02984","snapshot_observed_at":"2026-08-06T19:31:19.171471Z","title":"Cabannes, E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.171471Z"},"links":{"cited_paper":"/paper/2310.02984","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:d2876b059f11937745acb8acfd7a0fdc12d4c79c574dc26cf356c272f0dfad95","observation_id":"b81b91af-df0c-4091-a82e-b4c499256fed","resolution":{"observed_at":"2026-08-06T19:31:19.171471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18724","last_updated":"2024-02-28T21:47:30Z","snapshot_observed_at":"2026-08-04T18:39:27.530283Z","submitted_at":"2024-02-28T21:47:30Z","title":"Learning Associative Memories with Gradient Descent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18724","snapshot_observed_at":"2026-08-06T19:31:19.335976Z","title":"Cabannes, B","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.335976Z"},"links":{"cited_paper":"/paper/2402.18724","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:34803819aa1b3a8b337452652a62b464c3263f6e5669468f0d9a29a7e34e2648","observation_id":"94ed4ca1-0948-41bf-b368-c8583e1764af","resolution":{"observed_at":"2026-08-06T19:31:19.335976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.936391Z","title":"Damian, J","venue":null,"work_id":"4d3b54fa-c487-46be-95dd-cc37cadfbbe6","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.459944Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:fff28808d4ec951826e3bf4ab1bff6223adbeed9d30f4a41090cf79957411a25","observation_id":"364be1d2-467b-4b3d-ba35-6b2fd9234541","resolution":{"observed_at":"2026-08-06T19:31:24.940650Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.921714Z","title":"Davis and W","venue":null,"work_id":"b95ec9f8-636e-4c10-b24b-75191d6ed88f","year":1970},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.606152Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:a037e175856e3c0093c389a4edba29dcf7107c71ea805e861a422ceeea7051d7","observation_id":"a08c7306-3f91-4f51-9c8d-7d5767e9cbf9","resolution":{"observed_at":"2026-08-06T19:31:24.926141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11004","last_updated":"2024-02-16T18:28:36Z","snapshot_observed_at":"2026-08-03T19:49:17.673100Z","submitted_at":"2024-02-16T18:28:36Z","title":"The Evolution of Statistical Induction Heads: In-Context Learning Markov Chains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11004","snapshot_observed_at":"2026-08-06T19:31:19.725776Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.725776Z"},"links":{"cited_paper":"/paper/2402.11004","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:a5df57560649c33df68942b8ef8a638335f728b74d891ef3633776a790a797ba","observation_id":"ddb929c6-eeb2-4083-b98d-6709255ca34d","resolution":{"observed_at":"2026-08-06T19:31:19.725776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.906805Z","title":null,"venue":null,"work_id":"93d61bbc-8cf3-49ed-b55f-07f5bc01ca09","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:19.908449Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5c4a0157e908f106dc272b2f77babe680227a7b80bce0b945d27ff91c5d02c8a","observation_id":"2eea3257-3082-4f41-bcb0-bd63cafb9b10","resolution":{"observed_at":"2026-08-06T19:31:24.911625Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.892962Z","title":"Fernandez-Delgado, E","venue":null,"work_id":"28470f86-c177-4f60-aa55-eb41c17d0869","year":2014},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.081513Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5d0e95a33d2275e9a52c3edfa3c00aec56a4357cdcf8ca50c1db1e8657eb4c22","observation_id":"c4371282-5327-4e9f-90cc-ff919e938528","resolution":{"observed_at":"2026-08-06T19:31:24.897099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.878958Z","title":null,"venue":null,"work_id":"883af1ce-36af-455a-a702-30c51a2ca855","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.153867Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:1b4738051dc4900d7fe5417b5901efddd79bdaa3f762e041fae3d4e7c5ea1488","observation_id":"4c035b1c-aeb2-432e-bd61-0f68a23ebe64","resolution":{"observed_at":"2026-08-06T19:31:24.883272Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.05794","last_updated":"2024-10-18T21:32:39Z","snapshot_observed_at":"2026-08-06T01:24:27.374782Z","submitted_at":"2022-06-12T17:06:35Z","title":"SGD and Weight Decay Secretly Minimize the Rank of Your Neural Network","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.05794","snapshot_observed_at":"2026-08-06T19:31:20.239832Z","title":"Galanti, Z","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.239832Z"},"links":{"cited_paper":"/paper/2206.05794","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:f5643f2f00b1306e376f6a98f40e36817ad68accbbc106e8419cc6f8231e79e7","observation_id":"706d7da5-4244-4ea5-aa3d-85df53b06cf3","resolution":{"observed_at":"2026-08-06T19:31:20.239832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.862974Z","title":"Gan and T","venue":null,"work_id":"020aa732-ca99-4a0f-9aa1-aafe39875b8f","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.334218Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:8ff948260e0e0c1293eeed87b4160fecb7c255f9262e7c3b4693d1c35442e893","observation_id":"8f85cd05-c950-4fb7-8c4b-4d2df31c04a4","resolution":{"observed_at":"2026-08-06T19:31:24.867454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02679","last_updated":"2023-01-06T19:00:01Z","snapshot_observed_at":"2026-08-06T08:20:11.326697Z","submitted_at":"2023-01-06T19:00:01Z","title":"Grokking modular arithmetic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02679","snapshot_observed_at":"2026-08-06T19:31:20.404668Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.404668Z"},"links":{"cited_paper":"/paper/2301.02679","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:e9ac70ee0a4a53bc51d3428e318f113e15a0afbbc51234771a017a74231fdbe5","observation_id":"8b77646c-ee6e-4b99-89e7-2011278f6431","resolution":{"observed_at":"2026-08-06T19:31:20.404668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.847750Z","title":"Gunasekar, J","venue":null,"work_id":"f9d6c54e-1c98-4406-aad3-ab4799950b90","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.464036Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:e13ff09e4e06f2842213d4acb6b48b996bb00ba42b78452d9d4bf3ed865b0e5e","observation_id":"aa502f56-32bf-400a-b429-ac33cba44b8b","resolution":{"observed_at":"2026-08-06T19:31:24.852195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.833629Z","title":"Gunasekar, B","venue":null,"work_id":"730ba1df-162b-4c31-89e3-b729a70ac6bd","year":2017},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.524900Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:745b975c3241f1d91713512b62d10458f970ee67d2baf3cf43055aa4a7d58e96","observation_id":"cd7e5fd8-d43d-4135-8fbc-21f33ddc52a2","resolution":{"observed_at":"2026-08-06T19:31:24.837780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.02073","last_updated":"2022-05-10T01:55:43Z","snapshot_observed_at":"2026-07-06T11:15:46.517682Z","submitted_at":"2021-06-03T18:31:41Z","title":"Neural Collapse Under MSE Loss: Proximity to and Dynamics on the Central Path","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.02073","snapshot_observed_at":"2026-08-06T19:31:20.601467Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.601467Z"},"links":{"cited_paper":"/paper/2106.02073","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:dcb106e3539ae54a7c89c61d33de75e82f82bfddb4afab6b7343db39fbe88534","observation_id":"1f07f671-682f-432a-9c04-4f42f1180631","resolution":{"observed_at":"2026-08-06T19:31:20.601467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.819469Z","title":"Jacot, F","venue":null,"work_id":"6fb19b1b-8b5d-4cf1-9738-19c29e82c3f4","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.687675Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:229efab18672f309bab11380305537edb311dacc90e345e147f64c31496da1c2","observation_id":"b978ebb9-e0bd-47df-a37d-d1cf202c3168","resolution":{"observed_at":"2026-08-06T19:31:24.823514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.804511Z","title":"Ji and M","venue":null,"work_id":"5bc42675-bb72-48e4-aeda-d39cea43262b","year":2019},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.774219Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:6adea10ca0636442bda709e665746fc900d029941b4d167f7803e6e1db7085fe","observation_id":"908d1050-2f3c-4e82-8e92-abb7418a68c3","resolution":{"observed_at":"2026-08-06T19:31:24.808800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.789302Z","title":"Ji and M","venue":null,"work_id":"ea92526e-3e95-4249-a60c-5868c4cf3386","year":2020},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.899292Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:4f98ba471d1470524aada57b67072be912475ddaa37edb8696a135cdfa048d4d","observation_id":"3f861106-d511-430a-8734-bee57e9b49d0","resolution":{"observed_at":"2026-08-06T19:31:24.794164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04041","last_updated":"2023-04-11T06:11:14Z","snapshot_observed_at":"2026-08-04T02:44:28.334010Z","submitted_at":"2022-06-08T17:55:28Z","title":"Neural Collapse: A Review on Modelling Principles and Generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04041","snapshot_observed_at":"2026-08-06T19:31:20.976793Z","title":"Kothapalli","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:20.976793Z"},"links":{"cited_paper":"/paper/2206.04041","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:8b5168c3f617451840d724d43f8289ad438d27039b54fc9df035eb7bf3d3ce91","observation_id":"950a49b5-cdf4-445e-8e06-ded169d96f66","resolution":{"observed_at":"2026-08-06T19:31:20.976793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:21.064475Z","title":"Krizhevsky, G","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.064475Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:d5e9dcf4e0e9126ca18ace5cb87ff1675ac6cdc76267f7dbd69c280a423b0352","observation_id":"c16ca467-e6f9-411b-9775-550facc9b7a1","resolution":{"observed_at":"2026-08-06T19:31:21.064475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06110","last_updated":"2024-04-11T16:15:34Z","snapshot_observed_at":"2026-08-04T18:50:08.971653Z","submitted_at":"2023-10-09T19:33:21Z","title":"Grokking as the Transition from Lazy to Rich Training Dynamics","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06110","snapshot_observed_at":"2026-08-06T19:31:21.129096Z","title":"Kumar, B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.129096Z"},"links":{"cited_paper":"/paper/2310.06110","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:353250e8236c9d25c93b3093feb37d956b4b97df4abeaa0f4c7e46e7134cfb59","observation_id":"59989079-aa73-4069-bec2-1dbc2a1823e9","resolution":{"observed_at":"2026-08-06T19:31:21.129096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.763798Z","title":null,"venue":null,"work_id":"8f2ac8c9-2909-46a0-b40d-93d95ff5f643","year":1998},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.211221Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:de8f599460eb0e2a42528023acd1b1df4bc0bc582d8788bbb9d66e9a028f43bb","observation_id":"d29c9979-b953-4e38-a1bf-c955c02af362","resolution":{"observed_at":"2026-08-06T19:31:24.768034Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.10585","last_updated":"2021-12-13T17:43:52Z","snapshot_observed_at":"2026-07-06T11:11:45.392079Z","submitted_at":"2021-05-21T21:50:18Z","title":"Properties of the After Kernel","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.10585","snapshot_observed_at":"2026-08-06T19:31:21.269704Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.269704Z"},"links":{"cited_paper":"/paper/2105.10585","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:dfe693ff2a1225b81b75625037db4767fae0f310c5396e626565fdab1c53ff6c","observation_id":"c5225fb2-4470-4553-b909-8b34d624a417","resolution":{"observed_at":"2026-08-06T19:31:21.269704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.05890","last_updated":"2020-12-29T05:33:37Z","snapshot_observed_at":"2026-07-06T08:00:13.822686Z","submitted_at":"2019-06-13T18:52:00Z","title":"Gradient Descent Maximizes the Margin of Homogeneous Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.05890","snapshot_observed_at":"2026-08-06T19:31:21.355070Z","title":"Lyu and J","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.355070Z"},"links":{"cited_paper":"/paper/1906.05890","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:696da894e37245671bb6210c286f7a445ebc39b883c84f7d700c4b7c71667e88","observation_id":"2230e2cb-89dc-40ef-a786-ec6da92d6e98","resolution":{"observed_at":"2026-08-06T19:31:21.355070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.748310Z","title":"Mallinar, D","venue":null,"work_id":"47d40f84-a94c-4cdc-a539-4db03a8c5040","year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.505786Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:bea1bf44a6cf6da97f25c4c55b82a68418ed44963c7ad97ef33e08f0d63e1ab1","observation_id":"ec1e8e2a-d381-4382-90bc-cbca99002a97","resolution":{"observed_at":"2026-08-06T19:31:24.752906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.733236Z","title":"Marion and L","venue":null,"work_id":"7d3b001a-c105-4507-9141-2d5c31fec9d7","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.564272Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:3d01a8dd069471571cb72694b7df8773ce4adead196a03ad69fbfef1e346c2d0","observation_id":"e2ea96ef-6db6-42e7-a4ec-6f12e2b8c96e","resolution":{"observed_at":"2026-08-06T19:31:24.738055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.718778Z","title":null,"venue":null,"work_id":"a80e9549-4cf0-4c5c-89c4-60eb7b477187","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.649908Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:f515efc9e02246019f6bde9d06548f51fe92bef1cf55ad2cf0a5e13a231f2b5f","observation_id":"731e296b-ac57-4d00-8e8d-241a89762000","resolution":{"observed_at":"2026-08-06T19:31:24.722954Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07568","last_updated":"2024-02-19T17:59:29Z","snapshot_observed_at":"2026-07-06T16:46:57.403995Z","submitted_at":"2023-11-13T18:56:33Z","title":"Feature emergence via margin maximization: case studies in algebraic tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07568","snapshot_observed_at":"2026-08-06T19:31:21.714527Z","title":"Morwani, B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.714527Z"},"links":{"cited_paper":"/paper/2311.07568","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:ef7b45ea1a64fe3912979d58522e2e7ef271a18182335d1b7c2ab7a6b57e5610","observation_id":"b25d254f-e221-4b1f-9701-47174cb8b4a0","resolution":{"observed_at":"2026-08-06T19:31:21.714527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.05217","last_updated":"2023-10-19T21:25:32Z","snapshot_observed_at":"2026-08-02T10:20:00.635719Z","submitted_at":"2023-01-12T18:56:49Z","title":"Progress measures for grokking via mechanistic interpretability","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.05217","snapshot_observed_at":"2026-08-06T19:31:21.789558Z","title":"Nanda, L","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.789558Z"},"links":{"cited_paper":"/paper/2301.05217","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:cbffa8def76ca0e670e50965887647ae5780914076de634332df00afd34b92b2","observation_id":"a4cb04f7-868d-4719-8839-5557a2c008b6","resolution":{"observed_at":"2026-08-06T19:31:21.789558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14735","last_updated":"2024-08-13T15:45:37Z","snapshot_observed_at":"2026-07-06T17:34:07.737296Z","submitted_at":"2024-02-22T17:47:03Z","title":"How Transformers Learn Causal Structure with Gradient Descent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14735","snapshot_observed_at":"2026-08-06T19:31:21.883499Z","title":"Nichani, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:21.883499Z"},"links":{"cited_paper":"/paper/2402.14735","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:74f39bce4d57be0c2a4c5c5346eaea5f421eaccd6c5cfad3a283174f53e77a48","observation_id":"e6ad2f75-2630-4327-84e0-ed74ac02d5f6","resolution":{"observed_at":"2026-08-06T19:31:21.883499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-06T19:31:22.053685Z","title":"Olsson, N","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.053685Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:5c93b8d270ced911e6cb8201c6f6c205d64a7bd1b1792a8513d9a52c52612228","observation_id":"ecb45ebc-fede-4a06-9a48-ddf234580563","resolution":{"observed_at":"2026-08-06T19:31:22.053685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.704513Z","title":"Radhakrishnan, D","venue":null,"work_id":"acc8aec0-632a-4012-9687-275c784cab3e","year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.138439Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:1f080d853c9267f2d0c1fff2ecf6f76ed0cc4a971c1d05c57558bfdf0ed3ebd9","observation_id":"0ca2b527-e84b-4955-8065-e6132fc15d43","resolution":{"observed_at":"2026-08-06T19:31:24.708689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.690510Z","title":"Radhakrishnan, M","venue":null,"work_id":"75ebcf0d-e818-4a2c-b187-70aaa49ea2be","year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.239240Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:822bb0300f70e19334e908549b64f3edb7200fa80e73dfbd93833dfb9345413c","observation_id":"24902931-396d-4ebb-ae42-dfefc601ed95","resolution":{"observed_at":"2026-08-06T19:31:24.694679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.674898Z","title":null,"venue":null,"work_id":"601b52b6-c7e6-4569-8467-7a96af810a55","year":1969},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.355074Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:55e5e24f7bd05b8c58b02bc3f4b269a3a4f4fc91bd3b3cacc6557ea698c1446e","observation_id":"9a72e09f-4fe4-4ee2-98d2-82d05b4f194d","resolution":{"observed_at":"2026-08-06T19:31:24.679642Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.660584Z","title":null,"venue":null,"work_id":"fbf25fa4-81d2-4c69-a687-649f0a460df1","year":2014},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.438935Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:18306e7af245de491c0b300c0ff15cd9259fe2f9d5cae04da6e99c6b22a6a194","observation_id":"7bb330c8-4970-42e6-847e-da508dcfbfc8","resolution":{"observed_at":"2026-08-06T19:31:24.665209Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.645502Z","title":"Schölkopf","venue":null,"work_id":"f86957e3-aee1-49ce-8ddd-ea0bb6cd9226","year":2002},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.618654Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:47b02c64fec201fb288ab8f26a048579b111705198590eb92376b3a76813084f","observation_id":"fbf637da-08f3-4cf0-b627-d1e2759742b0","resolution":{"observed_at":"2026-08-06T19:31:24.650123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.627904Z","title":"Soudry, E","venue":null,"work_id":"9455cee2-8567-4c16-a927-deedc6c31b1a","year":2018},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.728679Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:999cc34790748f4bc66abdb70177331954a0272c1c5f43d6f31d1a8f99991fa8","observation_id":"fcaf784b-b920-43de-94da-14a76e9b50b3","resolution":{"observed_at":"2026-08-06T19:31:24.634050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.612620Z","title":"Stewart, F","venue":null,"work_id":"531816a2-6909-4218-bb2a-169eeca0d71c","year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.817942Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:0ff066fbfd4a62c3acfb2d451adae8a275e65edb8b0d7688e1e9179c0dd6e1d7","observation_id":"578f007b-5beb-46ed-ae21-ba15c00ceea1","resolution":{"observed_at":"2026-08-06T19:31:24.616876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.596085Z","title":"Thompson","venue":null,"work_id":"5de6b7e4-5d84-4dc9-afda-8fedbc1ac209","year":1976},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:22.979868Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:78a9e6b8ee33e8f85dcf1b59fd7bacfd78c22a13ff9f0e9c4f8d6c96fbe9ccf4","observation_id":"2fee3bd5-31e5-4c92-a55e-04e0eb049272","resolution":{"observed_at":"2026-08-06T19:31:24.600429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.582052Z","title":"Woodworth, S","venue":null,"work_id":"bf3aaa6d-7156-496e-bfcc-30ce0a885d09","year":2020},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.157017Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:b30c3392e0b138c28323e10c6ceb9748553cfafa4b816c1918ea14a6674f850c","observation_id":"620f0ca0-1e19-4cef-863c-21cb30acd2f2","resolution":{"observed_at":"2026-08-06T19:31:24.586352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:23.296676Z","title":"Zangrando, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.296676Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:eb0fb6ea6155461db7c1d379b7d5ac2b68fb1b41a1fd8906134a7012a9f7c29b","observation_id":"b304a0c9-3587-4c97-bdcc-9f5161999290","resolution":{"observed_at":"2026-08-06T19:31:23.296676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:23.400916Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.400916Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:a647a1a79386f3ac55a5e787e12647ba8b0261485ed6f8588eeb33aa8bcad9e4","observation_id":"cd342bf4-0045-4985-acf7-0e0c7e1e5b85","resolution":{"observed_at":"2026-08-06T19:31:23.400916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.04815","last_updated":"2024-06-06T03:57:32Z","snapshot_observed_at":"2026-07-06T15:40:00.975837Z","submitted_at":"2023-06-07T22:37:11Z","title":"Catapults in SGD: spikes in the training loss and their impact on generalization through feature learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.04815","snapshot_observed_at":"2026-08-06T19:31:23.560693Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.560693Z"},"links":{"cited_paper":"/paper/2306.04815","citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:d0d5df7c533de5e15aac3fac8b065190d796a02b852d2523518d3ecd3da9da84","observation_id":"af7e1ca8-b352-49da-a3f4-621f29fa83df","resolution":{"observed_at":"2026-08-06T19:31:23.560693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.568140Z","title":"Ziyin, I","venue":null,"work_id":"8a915652-1ba4-4855-970d-d2107ce7c743","year":2025},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.719685Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:255e436ba2872a915f854d808c1c08c3254b7dcc1baee0ab0a78f86172df0042","observation_id":"706c1eac-8e9a-47dd-beec-0b0a644a6a20","resolution":{"observed_at":"2026-08-06T19:31:24.572126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:31:24.552190Z","title":"Ziyin, B","venue":null,"work_id":"4042745e-86e7-4c4c-a0c9-7414e1a7baca","year":2022},"citing_paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T19:31:23.888184Z"},"links":{"citing_paper":"/paper/2507.05644"},"observation_digest":"sha256:483d6c170f88204bb2c1b4a5ba13321039d067fbd9b95235d8a06dd5ae5cd273","observation_id":"11db1100-1a4e-49fe-a9b0-9225509e8ed9","resolution":{"observed_at":"2026-08-06T19:31:24.557660Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.05644","last_updated":"2025-09-05T01:58:57Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T19:18:52.304504Z","submitted_at":"2025-07-08T03:52:48Z","title":"The Features at Convergence Theorem: a first-principles alternative to the Neural Feature Ansatz for how networks learn representations"},"reference_resolution":{"displayed":53,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":0,"verified_fuzzy":24},"total_outbound_references":53},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 53 of 53 outbound references and 1 inbound Pith citation observation for arXiv:2507.05644."}