{"as_of":"2026-08-07T00:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c4f1f5e772d8be34aa5453d9b733e0dba8be011b61e0868d02d73ba1da455a2a","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-10T11:57:18.311409Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.08170/citation-record","integrity":"/paper/2607.08170/integrity","json":"/paper/2607.08170/citation-record.json","paper":"/paper/2607.08170"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.142176Z","title":"Ainsworth, Jonathan Hayase, and Siddhartha S","venue":null,"work_id":"6769d917-a754-4d7c-8db0-894e03b56c1c","year":2023},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:3d8f2abde000f3b30328fb11c628f48d5a699505008c181b0ca297eaa48ad3ea","observation_id":"7fa65313-406d-4a39-981d-59a5075baa5b","resolution":{"observed_at":"2026-07-10T12:07:04.143545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.135003Z","title":"Pythia: A suite for analyzing large language models across training and scaling","venue":null,"work_id":"5138ec61-b100-46e2-9f0b-778d354fce34","year":2023},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:bee764bae8b24624d6e424059486db1e5b68e9a088dad23c608fa7e91c12b7fd","observation_id":"a0ea91b2-635a-4a70-8766-ec940537e774","resolution":{"observed_at":"2026-07-10T12:07:04.136553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.149017Z","title":"PIQA: reasoning about physical commonsense in natural language","venue":null,"work_id":"99d8d551-f58b-4a4e-94d4-f0c87c01d368","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:6c8353ca70cc133f654c068219737a98f4d767fa4ae68cd044432beaeafe0443","observation_id":"1028b221-3d5d-4ef5-a449-8a63ce43d71e","resolution":{"observed_at":"2026-07-10T12:07:04.150449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.140262Z","title":"LL amaflex: Many-in-one LLM s via generalized pruning and weight sharing","venue":null,"work_id":"156c8367-92f1-4bac-bef0-070ec85c16f1","year":2025},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:cba83a3d74d2d49aec5f87fb1befbda5674391b90ed8f2981adbd17c71f5ee66","observation_id":"f9ed2c96-574f-4cb0-932b-d567d9f437b8","resolution":{"observed_at":"2026-07-10T12:07:04.141686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n19-1300","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"B ool Q : Exploring the Surprising Difficulty of Natural Yes/No Questions","venue":null,"work_id":"b0eff16f-bbcd-4d66-a41d-d89ff07a80e5","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:4c93f1e81b299a24c976c76c00924a13025dcd128890833b7fd15bd29b0d1570","observation_id":"00801676-17bb-4cbb-86e9-fb79b2a77fd8","resolution":{"observed_at":"2026-07-10T12:07:03.406159Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T15:19:26.94061+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T15:19:26.94061+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":"1803.05457","doi":"10.1162/tacl_a_00448.https://aclanthology.org/2022.tacl-1.5","metadata_source":"pith","pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","venue":"cs.AI","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","year":2018},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:03dda9bb6e27568ea727a58c527422c6ca471b0ed8788ac669eab5caa5e97a7e","observation_id":"c609ea54-dbef-4602-9f80-fe77aabb2a95","resolution":{"observed_at":"2026-07-10T12:07:03.715577Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-04T15:46:25.710484Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":"2110.14168","doi":"10.1002/j.1545-","metadata_source":"pith","pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Training Verifiers to Solve Math Word Problems","venue":"cs.LG","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","year":2021},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:572c08df0d511e0029e764c88691d57f1833847e025a974bef8277868ed69b73","observation_id":"337f662d-d850-45c0-a642-c4fa805e5ccf","resolution":{"observed_at":"2026-07-10T12:07:03.728260Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/n19-1423","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BERT : Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":"Proceedings of the 2019 Conference of the North","work_id":"3e3c8ac8-b858-4b22-af32-393d98c883e0","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:e483b4b714357b0f337031959c8b4d5be9fd96591ff7f79491ea5aae58e71c0f","observation_id":"d5a81ca7-9983-49fe-aa9d-9bb79ee8dfab","resolution":{"observed_at":"2026-07-10T12:07:03.419449Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T13:38:13.85894+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T13:38:13.85894+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.06296","last_updated":"2022-07-05T11:40:49Z","snapshot_observed_at":"2026-08-06T06:25:19.456607Z","submitted_at":"2021-10-12T19:28:48Z","title":"The Role of Permutation Invariance in Linear Mode Connectivity of Neural Networks","version":2},"cited_work":{"arxiv_id":"2110.06296","doi":null,"metadata_source":"pith","pith_arxiv_id":"2110.06296","snapshot_observed_at":"2026-07-11T01:17:45.386738Z","title":"The role of permutation invariance in linear mode connectivity of neural networks.arXiv preprint arXiv:2110.06296","venue":"cs.LG","work_id":"6eb08d4c-aaf2-49d4-9c48-c12ea6220af0","year":2021},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2110.06296","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:c7e45469967e5a851602d617a81d00907210f153e26528de8b1dc5e7678f46e8","observation_id":"4df61d97-ead5-41cf-9f09-08c143b108f0","resolution":{"observed_at":"2026-07-10T12:07:03.721928Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.138580Z","title":"Linear mode connectivity and the lottery ticket hypothesis","venue":null,"work_id":"8085a994-9e72-4ae7-b7f1-4e8afdae7f24","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:a78ecfd459a8e5783054c534e9365058ae342ae487430c58ea0d767bf7825181","observation_id":"4f52c028-f2d0-41e2-a701-3bfb560b9886","resolution":{"observed_at":"2026-07-10T12:07:04.140126Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00027","last_updated":"2020-12-31T19:00:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-12-31T19:00:10Z","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","version":1},"cited_work":{"arxiv_id":"2101.00027","doi":"10.1117/1.jmi.10.6.061104","metadata_source":"pith","pith_arxiv_id":"2101.00027","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","venue":"cs.CL","work_id":"9b10667a-da61-4358-aceb-10578234d45d","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2101.00027","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:1e5734e069a70b6b6e2cb2557918f13676923beb272dbe9854dc204664ad0561","observation_id":"916bff3e-c7ba-40b0-b11a-10b879f95800","resolution":{"observed_at":"2026-07-10T12:07:03.692701Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"records/1025683","doi":"10.5281/zenodo.10256836.url:https://zenodo.org/records/10256836","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Ariel Gera, Odellia Boni, Yotam Perlitz, Roy Bar-Haim, Lilach Eden, and Asaf Yehudai","venue":null,"work_id":"1a8adc3d-c761-4a48-acc1-c04d569c2d4a","year":2023},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:f982117d54010cba8b7f708582f7ae7c9d835445f5787b9e0ce122009a08a2fe","observation_id":"5c4da880-6d64-41af-a677-869b52d8132e","resolution":{"observed_at":"2026-07-10T12:07:03.689822Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:1c72192a0d8cd802dcc57073eb311fd6614900a70f6f2c170c9ccc6c3fb4b774","observation_id":"753d90a8-18aa-4def-b467-59b20b1889bc","resolution":{"observed_at":"2026-07-10T12:07:03.698716Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.172277Z","title":null,"venue":null,"work_id":"417141c6-c6c0-4969-ab1b-1fc95b23c0c9","year":2015},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:ad21c1efdea79413e8efad3eee053299f9299856e5bc4b6c54fbe1333ebb8ea9","observation_id":"a28d987c-d547-4b51-8c11-3097845455d2","resolution":{"observed_at":"2026-07-10T12:07:04.173578Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.170518Z","title":"Amc: Automl for model compression and acceleration on mobile devices","venue":null,"work_id":"f8af5604-bfef-4477-9b59-5970b2ae9291","year":2018},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:1cfa70b99d9116ee50ae418f093772ee5b6dda2c17396bd60525c825afea5a62","observation_id":"396b39c4-fd39-4290-aa70-0aec6d4b60b5","resolution":{"observed_at":"2026-07-10T12:07:04.171845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.164191Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":"2a098e3e-d84e-480a-a254-4fac859b95d6","year":2021},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:28bee0820bfac9852e1e32423a7d04e41c871be13a6a6cb4a61aa69edc1780b1","observation_id":"e80836d8-51da-4e68-a284-87ea425abb64","resolution":{"observed_at":"2026-07-10T12:07:04.165695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.159696Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":"86584887-5000-4659-93bb-62ca3f5358c7","year":2021},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:e34087bb75851b1d4d84e08139ff650c183c6f36a59a28dd3238f506ea674311","observation_id":"96cdab90-f6cf-468e-b328-f376721b2003","resolution":{"observed_at":"2026-07-10T12:07:04.161740Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.145462Z","title":"Rethinking layer relevance in large language models beyond cosine similarity","venue":null,"work_id":"d273eb61-8a0e-4954-aa21-1127cff9d8f1","year":2026},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:3d1e55ab6280b646b0e85e3ae6698419aeba2ab22856fd3b7d2ea7a9383cf855","observation_id":"0d3c99bc-ccd2-4db1-9db5-f38dbb6eb4da","resolution":{"observed_at":"2026-07-10T12:07:04.147020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":"1503.02531","doi":"10.1109/cvpr52733.2024.01515","metadata_source":"pith","pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Distilling the Knowledge in a Neural Network","venue":"stat.ML","work_id":"d927ab1f-17b8-4002-9d09-c3d55764fbad","year":2015},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:79bb34d939c0673b010f9a690ff7497618d1b5f7ba981768b4dce0509ec459c9","observation_id":"5df3f66f-e3c1-411e-a3f8-a0262a8629ef","resolution":{"observed_at":"2026-07-10T12:07:03.683139Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-07-06T12:54:11.616335Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":"2203.15556","doi":"10.1098/rsta.2024.0522","metadata_source":"pith","pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Training Compute-Optimal Large Language Models","venue":"cs.CL","work_id":"b2faf28d-86b7-429c-bc42-469458efc246","year":2022},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:4c0d28ee00a63d6ee54456660a848615410769ce32c7b1c1f822f453bd6b0a50","observation_id":"966089bb-212f-43c4-b41c-a9395960110d","resolution":{"observed_at":"2026-07-10T12:07:03.712525Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.02086","last_updated":"2025-06-02T21:33:43Z","snapshot_observed_at":"2026-08-02T20:26:29.728788Z","submitted_at":"2025-01-03T20:19:14Z","title":"Instruction-Following Pruning for Large Language Models","version":3},"cited_work":{"arxiv_id":"2501.02086","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.02086","snapshot_observed_at":"2026-07-10T12:07:03.716884Z","title":"arXiv:2501.02086 [cs]","venue":"cs.CL","work_id":"02691984-32c8-421a-9573-d3d1b1f3334e","year":2025},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2501.02086","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:8ed0f25de18694a4e6c63d95f5d8f3393625afdf71572d6e93645ca76594cfb9","observation_id":"f87028e6-38f2-49b8-aa6c-3f5df5b83b12","resolution":{"observed_at":"2026-07-10T12:07:03.718737Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.136698Z","title":"O'Reilly Media, Inc","venue":null,"work_id":"d2658dfb-b54a-45e9-8f16-4c1b747cd399","year":2022},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:75cdf624060c09328bb9b3097ef52e3ce5f710e3b505e10c26df9d40bde4ed0f","observation_id":"955e1785-d3d3-4122-b737-2a6bebb2019a","resolution":{"observed_at":"2026-07-10T12:07:04.137997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.129165Z","title":"Nayak, Jonathan Geuter, Marco Fumero, Francesco Locatello, and David Alvarez-Melis","venue":null,"work_id":"cdefa623-8238-49b7-b8b4-bb0038cd9b61","year":2026},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:9436a9319f235dc9e6f7736eeecd667be90b36a63edad83428dfcba98c3f6b2d","observation_id":"e967fbe6-0336-40d8-942c-1efeeb80220f","resolution":{"observed_at":"2026-07-10T12:07:04.130712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":"2001.08361","doi":"10.1145/3616855.3635845","metadata_source":"pith","pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling Laws for Neural Language Models","venue":"cs.LG","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:d1b9ab15ebd15005cce45ad05205520377afdb71353269d1fc28bf060b526b32","observation_id":"3a864d06-6e24-4d30-b641-ad7edff167cb","resolution":{"observed_at":"2026-07-10T12:07:03.680095Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.127302Z","title":"Nayak, Jack Merullo, Stephen Bach, Chen Sun, and Ellie Pavlick","venue":null,"work_id":"87799495-8916-49de-bcf6-7d806623498e","year":2025},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:ee440a00a977327b710d7d7a120509484b44d0c18b1d2308e04024869cb85270","observation_id":"81037bef-1336-407a-a0a6-79822c3263b6","resolution":{"observed_at":"2026-07-10T12:07:04.128645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/d17-1082","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.18653/v1/D17-1082","venue":null,"work_id":"4e65f57b-0562-4172-b6d3-b4ea2e5e62fb","year":2017},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:566c0b92d0e7215457c3083e69c4abc9d563291d913f6dc1b96690827d57943c","observation_id":"71f85d1c-61c9-435e-99dd-c373e392dcda","resolution":{"observed_at":"2026-07-10T12:07:03.414162Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T23:49:40.495585+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T23:49:40.495585+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.123617Z","title":"Optimal brain damage","venue":null,"work_id":"b0752379-0f14-4e09-b49a-62f5c51d8fc5","year":1989},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:81e9646acbf61352d5eb945cd0d3c63b492b700cc299b0d65f00eff32752a48f","observation_id":"7014a3f6-b773-4718-9339-e4472f48ca7d","resolution":{"observed_at":"2026-07-10T12:07:04.125328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.14199","last_updated":"2022-10-25T17:45:36Z","snapshot_observed_at":"2026-07-06T14:10:20.983444Z","submitted_at":"2022-10-25T17:45:36Z","title":"Same Pre-training Loss, Better Downstream: Implicit Bias Matters for Language Models","version":1},"cited_work":{"arxiv_id":"2210.14199","doi":"10.48550/arxiv.2210.14199","metadata_source":"pith","pith_arxiv_id":"2210.14199","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Same Pre-training Loss, Better Downstream: Implicit Bias Matters for Language Models","venue":"cs.LG","work_id":"26048d92-7f9a-45f5-98ff-a6a7fba430ac","year":2022},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2210.14199","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:7e04cbb9099106ddaed59864dbe4d215616c79889e62ebadfa3c352771822a35","observation_id":"46aeb89d-b959-4661-92ee-299d1e2954e0","resolution":{"observed_at":"2026-07-10T12:07:03.403700Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.167570Z","title":null,"venue":null,"work_id":"4a205cb8-7f5e-4854-ae2f-04aa4c766175","year":2025},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:2793a42dc17971955560cb2df8cfc64b0aad4069653cdabd78d77e059873336c","observation_id":"11a21545-5c0f-46a2-b918-614ac0053b22","resolution":{"observed_at":"2026-07-10T12:07:04.168909Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03853","last_updated":"2024-10-11T09:43:32Z","snapshot_observed_at":"2026-08-03T01:48:18.654934Z","submitted_at":"2024-03-06T17:04:18Z","title":"ShortGPT: Layers in Large Language Models are More Redundant Than You Expect","version":3},"cited_work":{"arxiv_id":"2403.03853","doi":"10.48550/arxiv.2403.03853","metadata_source":"pith","pith_arxiv_id":"2403.03853","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Shortgpt: Layers in large language models are more redundant than you expect","venue":"cs.CL","work_id":"195a45aa-5b87-4058-97eb-f08eca8ee8c6","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2403.03853","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:9c39422d505e2fce92720b0c628646ce634d0681dc85f51eb5a198388dd38f6d","observation_id":"90b12f9b-e33d-4cf7-b36a-003f54b7a14f","resolution":{"observed_at":"2026-07-10T12:07:03.709115Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.165858Z","title":"Pointer sentinel mixture models","venue":null,"work_id":"2d20f0cc-14f0-4ebe-a9ce-7ef84878b471","year":2017},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:a57ce595b5174e972286e403a6172c7038a4c9be474b34c159a5f57a057ab1dd","observation_id":"c0c83fa9-16d7-4459-97ad-9f410b59bdc7","resolution":{"observed_at":"2026-07-10T12:07:04.167239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/d18-1260","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering","venue":null,"work_id":"642f5867-b098-4ec2-9fd0-c6dd37388412","year":2018},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:26cb79183f2ab5461114e4351298ea6952dac3cf3c338809436d6838237647ad","observation_id":"910130a8-9205-4a55-a587-01cf979e16ec","resolution":{"observed_at":"2026-07-10T12:07:03.399735Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-17T19:51:13.783624+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-17T19:51:13.783624+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"hash/4822991","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:03.729103Z","title":"Compact language models via pruning and knowledge distillation","venue":null,"work_id":"33615907-8473-429f-95b3-8a836bf57fb2","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:b5fac6f6fa0809e533fdfbfc0a65782fae872837eac8931caca2a68c0b03c3c1","observation_id":"e43197a6-6d1e-46d7-a9d5-460d6d5cbbd0","resolution":{"observed_at":"2026-07-10T12:07:03.730788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.161992Z","title":"Uniform convergence may be unable to explain generalization in deep learning","venue":null,"work_id":"6c257b4e-297c-401f-ab75-2e51ae83862b","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:679fdff6770edd9ae216fec4b757a9914ee0b9f98ea4ae3d281ae15f7c63c429","observation_id":"7b132166-c204-4744-93b2-d2d79b2bc26c","resolution":{"observed_at":"2026-07-10T12:07:04.163433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/bf01588971","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:03.414861Z","title":"Rodrigo Nogueira, Zhiying Jiang, Ronak Pradeep, and Jimmy Lin","venue":"Mathematical Programming","work_id":"e0e25147-c540-4cc7-bb69-2d244d0ca62d","year":1978},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:f1d0aff043662a175678150793293b9b0ccdf3ee3fbf17da56b96123510d93a7","observation_id":"c8d607ee-732a-49f1-80d1-4a60ab499cf0","resolution":{"observed_at":"2026-07-10T12:07:03.416931Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T09:49:32.257461+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T09:49:32.257461+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.157709Z","title":"Interpreting gpt: The logit lens, 2020","venue":null,"work_id":"65f703f6-fdad-43d5-a1da-b0303fccef4e","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:cf3f79e1b157069326fd34e186f1d179fc4ab528420a9d407cf5dd1844c242b3","observation_id":"1749a420-2d00-45ca-bdac-d9112fb30bda","resolution":{"observed_at":"2026-07-10T12:07:04.158984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.156033Z","title":"Pytorch: An imperative style, high-performance deep learning library","venue":null,"work_id":"5d549792-c2d6-4b94-baf4-7367cfe21f1c","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:c789f820bd0324438304e5ab97a441792530cd281b3eacb48c034395859ac082","observation_id":"d9efb753-9221-4597-9ff3-bf79d0fc9070","resolution":{"observed_at":"2026-07-10T12:07:04.157590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.154120Z","title":"Language models are unsupervised multitask learners","venue":null,"work_id":"3af5d683-055b-4aee-b372-94d05b0ff627","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:97a0e892f681b65a086d8b32fe67341f90b71082263980b9ca99b481f50ecd82","observation_id":"f467c4d1-40a4-4c87-9fbc-3761538652e1","resolution":{"observed_at":"2026-07-10T12:07:04.155515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.152425Z","title":"Winogrande: An adversarial winograd schema challenge at scale","venue":null,"work_id":"47d01d49-aa45-459d-a326-3b41a386ba98","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:90c2210f098b3a0355a1cd355fbce9385d1be83539ae2317996a395b900f75c6","observation_id":"7bb48c76-fe0e-448c-938d-e0971d15cf0a","resolution":{"observed_at":"2026-07-10T12:07:04.153986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01108","last_updated":"2020-03-01T02:57:50Z","snapshot_observed_at":"2026-08-03T03:32:02.223426Z","submitted_at":"2019-10-02T17:56:28Z","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","version":4},"cited_work":{"arxiv_id":"1910.01108","doi":"10.48550/arxiv.1910.01108","metadata_source":"pith","pith_arxiv_id":"1910.01108","snapshot_observed_at":"2026-07-11T02:57:46.620845Z","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","venue":"cs.CL","work_id":"756f9764-ecd6-4672-8043-b37c698c7ad2","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/1910.01108","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:3ffbeb18ee44918be49ec0fb02e2096fa4e52af2aa478e52420481f4449c5fc9","observation_id":"818d213d-fd37-4689-8ec3-c77440f943f6","resolution":{"observed_at":"2026-07-10T12:07:03.733882Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.143736Z","title":"Model fusion via optimal transport","venue":null,"work_id":"e56a788e-47df-44b9-b48a-de74f44f8a3f","year":2020},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:b482203a8eccdfc059a19b65bbb2246e3dc8e73b7ad61368b0feb31016ee0f8d","observation_id":"c9366e22-2b15-4b23-85d0-cff50a68edaf","resolution":{"observed_at":"2026-07-10T12:07:04.145067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.11796","last_updated":"2024-12-09T18:31:01Z","snapshot_observed_at":"2026-08-04T02:12:19.734094Z","submitted_at":"2024-08-21T17:38:48Z","title":"LLM Pruning and Distillation in Practice: The Minitron Approach","version":4},"cited_work":{"arxiv_id":"2408.11796","doi":"10.48550/arxiv.2408.11796","metadata_source":"pith","pith_arxiv_id":"2408.11796","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llm pruning and distillation in practice: The minitron approach","venue":"cs.CL","work_id":"05faf480-8ccd-499e-af9e-dddde50a7987","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2408.11796","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:142ea731be172c9f2e9dc2f452a06ae36158427c715a95d7731204fe666db3c4","observation_id":"877f90ef-a88a-4e85-a253-d2047d6b52b8","resolution":{"observed_at":"2026-07-10T12:07:03.725405Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.150678Z","title":"Zico Kolter","venue":null,"work_id":"782611a2-52b0-41e1-ad6a-707575c6e00c","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:aa48015357cc89322eeb21ef0fd2a2e1ad93e16ae0c6c544dd89a4f1c92142b0","observation_id":"8caf3e99-830b-426d-bdcf-3042d6ce40f8","resolution":{"observed_at":"2026-07-10T12:07:04.152021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.147179Z","title":"The bitter lesson","venue":null,"work_id":"06c0884f-b45f-4768-934b-14b69bf3adac","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:a99985af6f7223492762153488e62e55cf72cc5bdb23f8f3878c80652c3de5b0","observation_id":"0c607b40-756c-42cc-886f-327b08e4421b","resolution":{"observed_at":"2026-07-10T12:07:04.148406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":"2602.02276","doi":"10.48550/arxiv.2602.02276","metadata_source":"pith","pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2.5: Visual Agentic Intelligence","venue":"cs.CL","work_id":"d690be8f-5d53-49b0-b1e7-79668eb8fcdb","year":2026},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:f77292c45af6b0e8b5e6f42804b63744de413c72cd09a0382a296941559af40d","observation_id":"a602ef30-b476-4a5e-bae9-30870e69b951","resolution":{"observed_at":"2026-07-10T12:07:03.705479Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.133294Z","title":"Qwen3.5: Accelerating productivity with native multimodal agents, February 2026","venue":null,"work_id":"69bba8c6-3bf5-4659-8e8c-554621837811","year":2026},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:cddc41d177e38c63d6cdd8cb2150a7941571600b08823ee3c5121537c46e2ea0","observation_id":"01e7f9c9-dd72-4663-8c18-dcb88f716180","resolution":{"observed_at":"2026-07-10T12:07:04.134828Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.130837Z","title":null,"venue":null,"work_id":"5b3f0401-8d41-4a54-8039-aba9048929ec","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:2e26954f9a5dade8556fab507c4f5043606c6502f827f81af3bd1997da018c1c","observation_id":"56f614f1-a074-4f79-a4f5-97883e2dea78","resolution":{"observed_at":"2026-07-10T12:07:04.132064Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.03771","last_updated":"2020-07-14T03:42:34Z","snapshot_observed_at":"2026-07-06T08:27:58.343233Z","submitted_at":"2019-10-09T03:23:22Z","title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","version":5},"cited_work":{"arxiv_id":"1910.03771","doi":"10.48550/arxiv.1910.03771","metadata_source":"pith","pith_arxiv_id":"1910.03771","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","venue":"cs.CL","work_id":"9d86da8d-01d3-41af-a0d2-ee14897927a9","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/1910.03771","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:7079157dd513c9c6a799592eb700bcaf13c6e4e171d674aa201ccb5d2db981d3","observation_id":"42051977-8d06-48b6-808a-b3dcbc7f7dc0","resolution":{"observed_at":"2026-07-10T12:07:03.695725Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:04.125559Z","title":"Sheared llama: Accelerating language model pre-training via structured pruning","venue":null,"work_id":"6130e8a6-5796-4dd9-bbb6-df29f4851132","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:e0a9bb5711d2ad06f7eb97a2f11ee0b1f6b476db6ee52da2c5438e64f627d09f","observation_id":"02afb929-8c0b-43ed-86a9-c2fa3849fb7d","resolution":{"observed_at":"2026-07-10T12:07:04.127141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:bfd4ef95e50f120403025e7b6bd7099ed72445fbd8a9852cb52705bac6a2d2ec","observation_id":"b494bd0b-3967-4ac7-8abc-bad887e620f5","resolution":{"observed_at":"2026-07-10T12:07:03.702530Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3787849","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Model merging in llms, mllms, and beyond: Methods, theories, applications, and opportunities","venue":"ACM Computing Surveys","work_id":"df41fd58-dcdd-4884-8624-77e25cea1bf8","year":2026},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:68e7beb5047a446bebb0e18c76d137d1b3f4f074b6c14298beb119f07737ebdf","observation_id":"47529697-8b99-448c-92ac-cc6fe713462d","resolution":{"observed_at":"2026-07-10T12:07:03.411619Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/p19-1472","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:07:03.406813Z","title":"URL https:// doi.org/10.18653/v1/p19-1472","venue":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics","work_id":"11bfc949-547c-40f3-a86d-953eb9b2154c","year":2019},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:d99be799400aa3a1c02dabfcc66c4d8109ef3b7a3542b1119452e040c1a75341","observation_id":"f7c222e6-126e-4317-baa8-203ead158e93","resolution":{"observed_at":"2026-07-10T12:07:03.408743Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T09:08:05.023254+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T09:08:05.023254+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18218","last_updated":"2024-10-20T09:10:25Z","snapshot_observed_at":"2026-08-06T18:43:32.229808Z","submitted_at":"2024-05-28T14:21:15Z","title":"FinerCut: Finer-grained Interpretable Layer Pruning for Large Language Models","version":2},"cited_work":{"arxiv_id":"2405.18218","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.18218","snapshot_observed_at":"2026-07-10T12:07:03.684455Z","title":"FinerCut: Finer-grained Interpretable Layer Pruning for Large Language Models","venue":"cs.LG","work_id":"9916430b-6a14-499f-bf78-592ea79e1b8d","year":2024},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2405.18218","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:a5ddeea4cb0e64d79e994b7fba96e8e444f3d66ae8e9d3660722723837b8e954","observation_id":"3dc41cbf-6610-45b7-9cf8-fa08de4600f8","resolution":{"observed_at":"2026-07-10T12:07:03.686574Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":"2311.07911","doi":"10.48550/arxiv.2311.07911","metadata_source":"pith","pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction-Following Evaluation for Large Language Models","venue":"cs.CL","work_id":"3aa06177-125a-4f5a-8f4a-8070c5986c26","year":2023},"citing_paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-07-10T11:57:18.311409Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2607.08170"},"observation_digest":"sha256:f7905cec3b83666683b55b217652e5820b80f11227a5f4d71740dce632549e56","observation_id":"64a621ed-3d37-416d-bcdf-d08d691fb35c","resolution":{"observed_at":"2026-07-10T12:07:03.675069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.08170","last_updated":"2026-07-09T07:14:12Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T11:55:56.419446Z","submitted_at":"2026-07-09T07:14:12Z","title":"Understanding Layer Patching in Model Size Interpolation"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":3,"verified_exact":23,"verified_fuzzy":24},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 0 inbound Pith citation observations for arXiv:2607.08170."}