{"as_of":"2026-08-06T19:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:14597dca884452b44e2664063a7ef8f733b3d6c67a94bdb9e61334aea2ffa894","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T21:30:54.180653Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-25T05:55:10.325836Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-25T05:55:24.019027Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"cited_work":{"arxiv_id":"2602.20062","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2602.20062","snapshot_observed_at":"2026-07-01T02:17:19.837651Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","venue":null,"work_id":"4d8e5b33-fc2a-4f2a-ad75-d46f4b2aa37c","year":2026},"citing_paper":{"arxiv_id":"2605.20105","last_updated":"2026-05-19T16:56:56Z","snapshot_observed_at":"2026-07-06T23:30:44.092395Z","submitted_at":"2026-05-19T16:56:56Z","title":"Optimal Representation Size: High-Dimensional Analysis of Pretraining and Linear Probing","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T06:53:16.925588Z"},"links":{"cited_paper":"/paper/2602.20062","citing_paper":"/paper/2605.20105"},"observation_digest":"sha256:0ab2181395b4c6fbbbeab83d11a9815c07082e7dd0635324a28c60ce0eaac4e9","observation_id":"745444f6-ea4e-46ff-8ee8-50d251a35440","resolution":{"observed_at":"2026-07-01T02:17:19.837651Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"cited_work":{"arxiv_id":"2602.20062","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2602.20062","snapshot_observed_at":"2026-07-01T02:17:19.837651Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","venue":null,"work_id":"4d8e5b33-fc2a-4f2a-ad75-d46f4b2aa37c","year":2026},"citing_paper":{"arxiv_id":"2605.22972","last_updated":"2026-05-21T19:04:19Z","snapshot_observed_at":"2026-07-06T23:33:14.457450Z","submitted_at":"2026-05-21T19:04:19Z","title":"A mathematical theory of balancing relational generalization and memorization","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-25T05:55:10.325836Z"},"links":{"cited_paper":"/paper/2602.20062","citing_paper":"/paper/2605.22972"},"observation_digest":"sha256:f18bb644648390807a5095d4aa96086f3e4d96c5b8629f7ab454cad471501c2a","observation_id":"f93d44aa-e97b-45a6-8c3c-e82d51431e77","resolution":{"observed_at":"2026-07-01T02:17:19.837651Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2602.20062/citation-record","integrity":"/paper/2602.20062/integrity","json":"/paper/2602.20062/citation-record.json","paper":"/paper/2602.20062"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:49.884441Z","title":"Neural networks as kernel learners: The silent alignment effect, 10 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:49.884441Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ae136c1eb83cb2e0c85225879707cd09c4a05d56fd820ce192995a40c6d33629","observation_id":"59f0bb67-2507-4970-87db-f3892ad1695e","resolution":{"observed_at":"2026-08-02T21:30:49.884441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:49.916160Z","title":"M., Cholakkal, H., Shah, M., Yang, M.-H., and Khan, F","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:49.916160Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:5a3f97c439c1e02a53df8af51d581abb973731b75c5baf5242eae8d6b5dd18d0","observation_id":"8d0ee721-d2a7-4e30-901a-9730a05b89ad","resolution":{"observed_at":"2026-08-02T21:30:49.916160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:49.987592Z","title":"S., Woodworth, B","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:49.987592Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:1bac58108b9f59ae209b2755677d88c822793bf86e54da0a60c8e49addea01b3","observation_id":"3297e28c-2d48-4d67-b953-fdf13188f00a","resolution":{"observed_at":"2026-08-02T21:30:49.987592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.100342Z","title":"and Montanari, A","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.100342Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:f8cd51175e2a5ddfaef9866271516608e91930294736e26d183e1d6a578a0934","observation_id":"a16299bb-75e6-4c5b-9694-15f43c8e3e70","resolution":{"observed_at":"2026-08-02T21:30:50.100342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.209440Z","title":"and Montanari, A","venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.209440Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:6ffb031359e0a771fdcc40c649829baebdc7733c20273823ecafb9b1c40c0f40","observation_id":"d868dc36-ebcd-4177-854a-f6cf7a26873c","resolution":{"observed_at":"2026-08-02T21:30:50.209440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.268733Z","title":"and M \\\"u ller, R","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.268733Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:50c2ee725107231fbf4879ca3c1af70a314d4b03c5b4e5a93e9dbaaf58bb50fd","observation_id":"c5a0ad6c-a66f-41b8-98f5-5e4d34c5beb6","resolution":{"observed_at":"2026-08-02T21:30:50.268733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.313876Z","title":"R., and Schulz-Baldes, H","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.313876Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:328646dc9344a6e7adaed5be5478c0fb0a2eec1cd69f129a4dc7fc7ad4f2368c","observation_id":"1a85d9e5-1126-4fbe-84b9-7ecf1f339ec9","resolution":{"observed_at":"2026-08-02T21:30:50.313876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.353814Z","title":"Incremental learning in diagonal linear networks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.353814Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:9db9f0683a3743cc1e44130a7baf94115bad20a552129413376e683b0f9a4fd3","observation_id":"0f92376a-e77e-4f63-accc-5c40b8e580df","resolution":{"observed_at":"2026-08-02T21:30:50.353814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-02T21:30:50.405308Z","title":"On the opportunities and risks of foundation models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.405308Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4a0c849c7b40b5aaef2402d72c9d62f81a3b83de01f8e0eeaea4c1b562664938","observation_id":"95dbc523-b5af-490f-8cd6-81973723dfed","resolution":{"observed_at":"2026-08-02T21:30:50.405308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.492722Z","title":"Exact learning dynamics of deep linear networks with prior knowledge","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.492722Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:cddf2c3ef5446ab14c225bfd6254b31f912ed25206c13904846535b5055b9a2d","observation_id":"a9bd822d-ec60-4d37-873a-60dec6d26665","resolution":{"observed_at":"2026-08-02T21:30:50.492722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.594416Z","title":"and Bach, F","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.594416Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:6e07c79bce58958ce2058ae0166d0198e3aabf4f274cf97a1356bc25b041dad6","observation_id":"7a2baf06-380f-498c-b65c-b4dbbb39ae9e","resolution":{"observed_at":"2026-08-02T21:30:50.594416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:50.695889Z","title":"On lazy training in differentiable programming","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.695889Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:5a5e796032323a9dddd6e11b6f3e26be9c9afec4f6f726c8af5d021a3616d41d","observation_id":"3a2c5efa-99cf-474b-894c-aa759a367748","resolution":{"observed_at":"2026-08-02T21:30:50.695889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00194","last_updated":"2024-12-23T04:15:36Z","snapshot_observed_at":"2026-07-06T17:37:48.299877Z","submitted_at":"2024-02-29T23:46:28Z","title":"Ask Your Distribution Shift if Pre-Training is Right for You","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.00194","snapshot_observed_at":"2026-08-02T21:30:50.791331Z","title":"Ask your distribution shift if pre-training is right for you","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.791331Z"},"links":{"cited_paper":"/paper/2403.00194","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:c17b0d67c18a4ccd18bfee9771f7439092d60dc0f474725c21e33137f773c5be","observation_id":"d8a90f2f-58a3-470e-a3b1-6b9a35e89265","resolution":{"observed_at":"2026-08-02T21:30:50.791331Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.14623","last_updated":"2025-03-04T11:18:33Z","snapshot_observed_at":"2026-07-06T19:19:33.231002Z","submitted_at":"2024-09-22T23:19:04Z","title":"From Lazy to Rich: Exact Learning Dynamics in Deep Linear Networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.14623","snapshot_observed_at":"2026-08-02T21:30:50.893495Z","title":"C., Anguita, N., Proca, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:50.893495Z"},"links":{"cited_paper":"/paper/2409.14623","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:993ff931de1f9fcad9c81995c8c73c6d6a3ca5363dab81974054d41bcccc5d33","observation_id":"bb9fab7a-155a-40e6-b22b-bbfcdc86025d","resolution":{"observed_at":"2026-08-02T21:30:50.893495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.033052Z","title":null,"venue":null,"work_id":null,"year":1975},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.033052Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4bacbbc0ba1eba6e69311ac04295915d72ba756797a097286d45952950fe7e9d","observation_id":"006c9240-117a-46f7-b735-45e46987d7d1","resolution":{"observed_at":"2026-08-02T21:30:51.033052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.097585Z","title":"K., Paul, M., Kharaghani, S., Roy, D","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.097585Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:0145453b5d215d2cf3c911e862037bccd09ac72a51a42b43745fd56084c9bf40","observation_id":"76a65956-41f3-47e9-8680-0dae8e3b0dbf","resolution":{"observed_at":"2026-08-02T21:30:51.097585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.127385Z","title":"A theory of multineuronal dimensionality, dynamics and measurement","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.127385Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4d3697fb3eb59ccc6457d1bbc1cdd3455d0d8c2d6d8de0f7d23fa362f88d14e2","observation_id":"449667df-ff04-4aab-ae3d-870c14bdd642","resolution":{"observed_at":"2026-08-02T21:30:51.127385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.197288Z","title":"R., and Aoi, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.197288Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:1260eb1ebc56c2d59cc77cc3d63c1c6eaff1059ecc7af6fa97e956ed51155f3a","observation_id":"21a581be-8d00-40fb-94b9-e67b2ec89420","resolution":{"observed_at":"2026-08-02T21:30:51.197288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.243328Z","title":"Characterizing implicit bias in terms of optimization geometry","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.243328Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:621838308baf624f7fae492908f8ecd5637455725330dbfea6be5b8d0edda1fd","observation_id":"a714be6b-bd1d-4a54-b47a-0eb60c3b47c2","resolution":{"observed_at":"2026-08-02T21:30:51.243328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.288371Z","title":"and Verd \\'u , S","venue":null,"work_id":null,"year":1983},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.288371Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ba6f46316ad151174a94fa810e4a8e6b05c3afc241ff2ce250c9ddfe65b2b111","observation_id":"c5c43e10-ea5f-4ffb-9dc0-33ff57426479","resolution":{"observed_at":"2026-08-02T21:30:51.288371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1608.08614","last_updated":"2016-12-10T13:37:06Z","snapshot_observed_at":"2026-07-06T05:08:40.739115Z","submitted_at":"2016-08-30T19:45:09Z","title":"What makes ImageNet good for transfer learning?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1608.08614","snapshot_observed_at":"2026-08-02T21:30:51.355068Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.355068Z"},"links":{"cited_paper":"/paper/1608.08614","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:c1a807162b9e79a18a126d1bbc996e44c21bd8e045c9de0a7ada5cbebd19f0d7","observation_id":"55c7ae16-7424-40ce-a31a-a65023293add","resolution":{"observed_at":"2026-08-02T21:30:51.355068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.404512Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.404512Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:6d3f4f6076c1336528b0268cc87aea69ddf55dc4493736558c3a4aba22d207aa","observation_id":"38796d70-5e70-4226-9189-acd7406fa9f2","resolution":{"observed_at":"2026-08-02T21:30:51.404512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.439162Z","title":"Train on Validation (ToV): Fast data selection with applications to fine-tuning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.439162Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:d255e930d935ebf8b513275256044fda0f25d3076212c2206bafd3b851df7029","observation_id":"6e5ea2a4-2672-4f16-8bd4-25c9f88dae83","resolution":{"observed_at":"2026-08-02T21:30:51.439162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12786","last_updated":"2024-08-21T16:37:20Z","snapshot_observed_at":"2026-07-06T16:50:42.754354Z","submitted_at":"2023-11-21T18:51:04Z","title":"Mechanistically analyzing the effects of fine-tuning on procedurally defined tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12786","snapshot_observed_at":"2026-08-02T21:30:51.482772Z","title":"S., Dick, R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.482772Z"},"links":{"cited_paper":"/paper/2311.12786","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:17b8c7debe04eefc0622990689c8b250086441254f670eb2639a6355e02cd97c","observation_id":"222ab62a-eff7-478f-b07d-541fd0edfdff","resolution":{"observed_at":"2026-08-02T21:30:51.482772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.02774","last_updated":"2024-05-05T00:08:00Z","snapshot_observed_at":"2026-07-06T18:09:55.249620Z","submitted_at":"2024-05-05T00:08:00Z","title":"Get more for less: Principled Data Selection for Warming Up Fine-Tuning in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.02774","snapshot_observed_at":"2026-08-02T21:30:51.528872Z","title":"A., Sun, Y., Jahagirdar, H., Zhang, Y., Du, R., Sahu, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.528872Z"},"links":{"cited_paper":"/paper/2405.02774","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e7dff1d4b7adef5dabc275b8fd9a2f01d9db10d788f578f41a23843d2273da82","observation_id":"4dc8c455-ea4b-42ea-9e8c-799e53dfad81","resolution":{"observed_at":"2026-08-02T21:30:51.528872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.572721Z","title":"A., Xu, W., Avestimehr, A","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.572721Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:6eadba523dba779667a79da2f702737e07cdf55d61c2646e6757df46e8e199b0","observation_id":"dc8c2ec6-ab9b-4ab2-bf1e-6a43de119877","resolution":{"observed_at":"2026-08-02T21:30:51.572721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:51.614373Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.614373Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:fb19681e656425da6c855bbb92cb3b5d58012af9f36bfcba69026727b4ca38bf","observation_id":"08a4bd9c-c927-490c-9ea0-52ee5443f3ed","resolution":{"observed_at":"2026-08-02T21:30:51.614373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.10054","last_updated":"2022-02-21T09:03:34Z","snapshot_observed_at":"2026-08-04T02:30:25.691953Z","submitted_at":"2022-02-21T09:03:34Z","title":"Fine-Tuning can Distort Pretrained Features and Underperform Out-of-Distribution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.10054","snapshot_observed_at":"2026-08-02T21:30:51.714613Z","title":"Fine-tuning can distort pretrained features and underperform out-of-distribution","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.714613Z"},"links":{"cited_paper":"/paper/2202.10054","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:4f58811d0b8844afd48607c88d85ead5633650a0d9f8cc4a593cb1252dcfe7fc","observation_id":"22023ff0-97d8-4172-bf1d-f773f17e3571","resolution":{"observed_at":"2026-08-02T21:30:51.714613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06158","last_updated":"2024-10-12T21:38:28Z","snapshot_observed_at":"2026-08-06T16:04:26.423565Z","submitted_at":"2024-06-10T10:42:37Z","title":"Get rich quick: exact solutions reveal how unbalanced initializations promote rapid feature learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06158","snapshot_observed_at":"2026-08-02T21:30:51.868185Z","title":"Get rich quick: exact solutions reveal how unbalanced initializations promote rapid feature learning, 06 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.868185Z"},"links":{"cited_paper":"/paper/2406.06158","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:2b6825787a6d8914507951bea42ee8d4cc63c18412655a01dcf15d7dd7a43110","observation_id":"be810198-8210-43b1-879f-11a02e250506","resolution":{"observed_at":"2026-08-02T21:30:51.868185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1809.10374","last_updated":"2019-01-04T22:16:58Z","snapshot_observed_at":"2026-07-06T07:04:37.077852Z","submitted_at":"2018-09-27T06:47:58Z","title":"An analytic theory of generalization dynamics and transfer learning in deep linear networks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.10374","snapshot_observed_at":"2026-08-02T21:30:51.976081Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:51.976081Z"},"links":{"cited_paper":"/paper/1809.10374","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ba857320b73c93527aa1120555b7801a9fee2af2611bd5ccfba487fb816b37ef","observation_id":"4def346e-c0ed-42d8-85a5-c6bde3e50cd3","resolution":{"observed_at":"2026-08-02T21:30:51.976081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.128369Z","title":"and Lindsey, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.128369Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e2f70c462de54cdec7fa0aed0fc0e65cee26f78231d4903f7a75ae7a54cb3b78","observation_id":"92933dcf-3120-44e7-9713-58a2446c998f","resolution":{"observed_at":"2026-08-02T21:30:52.128369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1906.05890","last_updated":"2020-12-29T05:33:37Z","snapshot_observed_at":"2026-07-06T08:00:13.822686Z","submitted_at":"2019-06-13T18:52:00Z","title":"Gradient Descent Maximizes the Margin of Homogeneous Neural Networks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1906.05890","snapshot_observed_at":"2026-08-02T21:30:52.266398Z","title":"and Li, J","venue":null,"work_id":null,"year":1906},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.266398Z"},"links":{"cited_paper":"/paper/1906.05890","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:fadfe56b55434d5abf2265b3f3032d0e9496f97fe54eea7240ea0ffdd9997d97","observation_id":"3f06f2f7-2c4f-4e83-8e0d-6896c75f72d3","resolution":{"observed_at":"2026-08-02T21:30:52.266398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.404458Z","title":"A kernel-based view of language model fine-tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.404458Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:eefc184cdb7948eb426f0b078716743b5c2d7fc1ae5f01d1c17edbb49154a561","observation_id":"9431c584-f450-432a-abe9-9a46a7809795","resolution":{"observed_at":"2026-08-02T21:30:52.404458Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.571553Z","title":"Abide by the law and follow the flow: conservation laws for gradient flows, 12 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.571553Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e321d813fc2707af1a6b66c9f99f59c4544949ab07d0b7ad123f35e4d261479e","observation_id":"d9c035d6-40c7-4937-bdd4-7cd8522e6607","resolution":{"observed_at":"2026-08-02T21:30:52.571553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.693140Z","title":null,"venue":null,"work_id":null,"year":1987},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.693140Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:70ac1c4a1a00a7259fb9331920af14658ba29238f3384b525b34c23188248dd2","observation_id":"0c29b522-a8f3-4d71-8ca1-f5a0f933ceb3","resolution":{"observed_at":"2026-08-02T21:30:52.693140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1310.5479","last_updated":"2013-10-21T09:42:02Z","snapshot_observed_at":"2026-07-06T03:26:06.438010Z","submitted_at":"2013-10-21T09:42:02Z","title":"Applications of Large Random Matrices in Communications Engineering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1310.5479","snapshot_observed_at":"2026-08-02T21:30:52.805911Z","title":"R., Alfano, G., Zaidel, B","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.805911Z"},"links":{"cited_paper":"/paper/1310.5479","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:d33f721c743c523472f2ee4ea4c92a1c144d6801a7fd7973fb3f347a27a25c2c","observation_id":"93f1a70e-722c-4fa2-acf2-d6f997f13faf","resolution":{"observed_at":"2026-08-02T21:30:52.805911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.899019Z","title":"S., Gunasekar, S., Lee, J., Srebro, N., and Soudry, D","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.899019Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:d56ad4c2a434fced21a207e2b41050a209707a4e6aa36ef31431a5a367c992d4","observation_id":"bde92dd5-183e-47a9-8a75-8ee7c50c5330","resolution":{"observed_at":"2026-08-02T21:30:52.899019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:52.975413Z","title":"S., Ravichandran, K., Srebro, N., and Soudry, D","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:52.975413Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:771c8ec9350e9ee924c304d41f35dd85ba1834c6fc0d6a4b2f3a2e6a6c87eff7","observation_id":"c22e3cac-db5c-47f0-9256-e3f8d218b7ab","resolution":{"observed_at":"2026-08-02T21:30:52.975413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13296","last_updated":"2024-10-30T01:04:15Z","snapshot_observed_at":"2026-07-06T19:05:14.227812Z","submitted_at":"2024-08-23T14:48:02Z","title":"The Ultimate Guide to Fine-Tuning LLMs from Basics to Breakthroughs: An Exhaustive Review of Technologies, Research, Best Practices, Applied Research Challenges and Opportunities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13296","snapshot_observed_at":"2026-08-02T21:30:53.064503Z","title":"B., Zafar, A., Khan, A., and Shahid, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.064503Z"},"links":{"cited_paper":"/paper/2408.13296","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:a81c2265d1a89ceb54f24568e9f860283aefabdeafef21acfb8342e196750a54","observation_id":"f21c03c0-66a1-4bba-9af9-ef00efa212d4","resolution":{"observed_at":"2026-08-02T21:30:53.064503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.136876Z","title":"and Flammarion, N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.136876Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:03761b4a2f805591e1e6d91b5d6dd1f4baf137840d5ac8f6505c33dfe8f5a278","observation_id":"d851012b-3a44-4755-92f1-c5e1f9d2b4b1","resolution":{"observed_at":"2026-08-02T21:30:53.136876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.229231Z","title":"Implicit bias of sgd for diagonal linear networks: a provable benefit of stochasticity","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.229231Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:575b5f80d2ade3174185992dd55bc5319e6a1279676d1c8ff53b37501e07b2fa","observation_id":"d819b0ae-55b1-4cc3-a180-33aa0570a2d3","resolution":{"observed_at":"2026-08-02T21:30:53.229231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.323905Z","title":null,"venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.323905Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:36eb92f614f1c6eae850359ef46d373689fd57c53eced6561aac200dc4e75170","observation_id":"42bd3e56-58a5-40ed-96d2-94e6a7334b4c","resolution":{"observed_at":"2026-08-02T21:30:53.323905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.386758Z","title":"How do infinite width bounded norm networks look in function space? In Conference on Learning Theory, pp.\\ 2667--2690","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.386758Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e09878d3bd33ee3a8a0191fc5f5ed1b154b11d71964cea9b1ca42ad192fdd494","observation_id":"af03399a-9ae5-4531-9cc8-f0deb651bc0d","resolution":{"observed_at":"2026-08-02T21:30:53.386758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.456359Z","title":"L., and Ganguli, S","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.456359Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:2ab5386d9e60a853a9af12a7358963359b212a721b55f34b38c5404b3d553f89","observation_id":"21bbf31b-07a8-4f02-a7f4-ae55c7ab8873","resolution":{"observed_at":"2026-08-02T21:30:53.456359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.551434Z","title":"M., McClelland, J","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.551434Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:c3f40e7298d991c8bd5e7cbae93180afa96ca6468ed7503aff8e1df6bd059cba","observation_id":"02b0c49d-8e12-4dd6-a58d-8edafaa4affe","resolution":{"observed_at":"2026-08-02T21:30:53.551434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.658605Z","title":"A theoretical analysis of fine-tuning with linear teachers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.658605Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:abb8c4ab01b5aee9222ccf917c5090ab8d28672219a28b9579e48a12e22f0ffc","observation_id":"d87e80e0-6f31-4e46-8bbc-5cb87865e881","resolution":{"observed_at":"2026-08-02T21:30:53.658605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.766153Z","title":"S., Gunasekar, S., and Srebro, N","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.766153Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:5f32eb90eb15c3ab101886083f1968f91178aa096e709d3da1b262aeb69d634c","observation_id":"b9f7f493-3d3b-4472-ba5c-e4d11c9c1d5f","resolution":{"observed_at":"2026-08-02T21:30:53.766153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.08194","last_updated":"2025-07-07T20:57:29Z","snapshot_observed_at":"2026-07-06T19:31:23.072307Z","submitted_at":"2024-10-10T17:58:26Z","title":"Features are fate: a theory of transfer learning in high-dimensional regression","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08194","snapshot_observed_at":"2026-08-02T21:30:53.816520Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.816520Z"},"links":{"cited_paper":"/paper/2410.08194","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:26c82a96eefddd460c639986c6f3bda0662bed05ba90f550bdb6a2fc8222d926","observation_id":"1b790542-f1a9-42bf-bbc1-41d2f62b412d","resolution":{"observed_at":"2026-08-02T21:30:53.816520Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.872616Z","title":"and Sato, I","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.872616Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:ce1bb67266365057a25ae419b7f71c0ba1d6dde390af87c335422d167b9d8c5f","observation_id":"9a2fa254-11d8-406b-8bd3-4bc99f466981","resolution":{"observed_at":"2026-08-02T21:30:53.872616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:53.922347Z","title":"and Lu, W","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.922347Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:83b95ad6121b8a52db1e7dc598bbe4fa468935b6f833752ff30ce2e01f262219","observation_id":"17d811cc-42a8-4554-9269-ea3569dda412","resolution":{"observed_at":"2026-08-02T21:30:53.922347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.10012","last_updated":"2022-06-20T21:23:28Z","snapshot_observed_at":"2026-07-06T13:22:50.125014Z","submitted_at":"2022-06-20T21:23:28Z","title":"Limitations of the NTK for Understanding Generalization in Deep Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.10012","snapshot_observed_at":"2026-08-02T21:30:53.974565Z","title":"Limitations of the ntk for understanding generalization in deep learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:53.974565Z"},"links":{"cited_paper":"/paper/2206.10012","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:fbb89b19514db14ced9ff9e25c8c100cda3d07ab407acd5a173965615330c6b5","observation_id":"e20161cf-43e6-4be2-8c21-a002f48ac39d","resolution":{"observed_at":"2026-08-02T21:30:53.974565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:54.028680Z","title":"D., Moroshko, E., Savarese, P., Golan, I., Soudry, D., and Srebro, N","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.028680Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:48d8932398b49550d967f54934b023b6fe91bad25ecc4dd4941f0cdf988bd319","observation_id":"2a1577c9-56bd-489a-9557-235a30a3ba19","resolution":{"observed_at":"2026-08-02T21:30:54.028680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:54.086358Z","title":"How transferable are features in deep neural networks? Advances in neural information processing systems, 27, 2014","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.086358Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:aa02e016f3d3c138025834913463dc45275bc8a0070ae5e919bbe5d93f3a61e0","observation_id":"dbc813df-f6c7-4054-a144-af0f6cdfda0a","resolution":{"observed_at":"2026-08-02T21:30:54.086358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1611.03530","last_updated":"2017-02-26T19:36:40Z","snapshot_observed_at":"2026-08-01T16:56:59.989486Z","submitted_at":"2016-11-10T22:02:36Z","title":"Understanding deep learning requires rethinking generalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.03530","snapshot_observed_at":"2026-08-02T21:30:54.126249Z","title":"Understanding deep learning requires rethinking generalization","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.126249Z"},"links":{"cited_paper":"/paper/1611.03530","citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:e21216eacf4ad2cc713e933f55c9e0ee877093872dc465a41be3ce0d9579eab7","observation_id":"e137aa50-0b28-4426-a65f-3f5aec7f2996","resolution":{"observed_at":"2026-08-02T21:30:54.126249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T21:30:54.180653Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-02T21:30:54.180653Z"},"links":{"citing_paper":"/paper/2602.20062"},"observation_digest":"sha256:9efbd510d848a086ec3d14832f07a3946d50ff46b6d5edc4aa36fba1d496cb4d","observation_id":"37204e22-915a-4584-a8e6-0c62d39a36f1","resolution":{"observed_at":"2026-08-02T21:30:54.180653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.20062","last_updated":"2026-06-30T08:25:25Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T03:53:38.456812Z","submitted_at":"2026-02-23T17:19:33Z","title":"A Theory of How Pretraining Shapes Inductive Bias in Fine-Tuning"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":55,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 2 inbound Pith citation observations for arXiv:2602.20062."}