{"as_of":"2026-08-08T18:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4cbf437d8cc8deb7df7b0a7c35b593169595e2fe08fb332bc780b50971809202","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":25,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:05:15.374000Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":429,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"1909.11942","last_updated":"2020-02-09T03:00:18Z","snapshot_observed_at":"2026-07-06T08:24:44.631342Z","submitted_at":"2019-09-26T07:06:13Z","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","version":6},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T12:26:58.015594Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/1909.11942"},"observation_digest":"sha256:e2acc34aed1f10215a601ad0bd8d7a69c12cce17becca136df74e039a3baefbf","observation_id":"df108f19-e6b9-407a-9853-abd23d00e31d","resolution":{"observed_at":"2026-05-13T12:26:58.123489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"1910.01108","last_updated":"2020-03-01T02:57:50Z","snapshot_observed_at":"2026-08-07T19:07:36.327251Z","submitted_at":"2019-10-02T17:56:28Z","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","version":4},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-11T05:02:34.076464Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/1910.01108"},"observation_digest":"sha256:deb266037c7669c77babd9d9568c26bf39cf03ed4bac8f58bba9039acd45399b","observation_id":"1c9ae76d-aa94-46de-9d6f-6451b731c5b4","resolution":{"observed_at":"2026-05-11T05:02:34.182192Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2101.02388","last_updated":"2021-01-07T06:12:28Z","snapshot_observed_at":"2026-07-06T10:30:38.112588Z","submitted_at":"2021-01-07T06:12:28Z","title":"Knowledge Distillation in Iterative Generative Models for Improved Sampling Speed","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-17T03:56:19.837017Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2101.02388"},"observation_digest":"sha256:5dbe2a3eb4d27fdca0398cca13792e62e9bd25c1d24bf63430374644b0e78991","observation_id":"46bed6a4-fbd8-40fe-a3c1-6aa47f22cc38","resolution":{"observed_at":"2026-05-17T03:56:19.927545Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2210.08933","last_updated":"2023-02-14T06:45:49Z","snapshot_observed_at":"2026-07-06T14:06:34.310083Z","submitted_at":"2022-10-17T10:49:08Z","title":"DiffuSeq: Sequence to Sequence Text Generation with Diffusion Models","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-20T06:50:38.258692Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2210.08933"},"observation_digest":"sha256:779ab86eef259880eb8591cd7b58ef5f29ccc441ed5ec8c44a9ac067f2ec9903","observation_id":"f621dc27-ae8f-43be-b1bf-d9caff9b78c8","resolution":{"observed_at":"2026-05-20T06:50:38.320346Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2502.05075","last_updated":"2026-04-19T21:23:04Z","snapshot_observed_at":"2026-08-03T04:05:35.638196Z","submitted_at":"2025-02-07T16:46:43Z","title":"Discrepancies are Virtue: Weak-to-Strong Generalization through Lens of Intrinsic Dimension","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T03:24:07.851782Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2502.05075"},"observation_digest":"sha256:95906435b56de00cc1bfee9ef1189aee54894f8b32eb373dce3e803902fe44d6","observation_id":"5ec53084-21e3-433c-a1f3-684ac4d331f3","resolution":{"observed_at":"2026-05-23T03:25:20.537624Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-07T15:05:15.374000Z","title":"Well-read stu- dents learn better: On the importance of pre-training compact mod- els,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2505.16369","last_updated":"2025-05-27T06:49:39Z","snapshot_observed_at":"2026-08-07T22:08:02.323514Z","submitted_at":"2025-05-22T08:23:54Z","title":"X-ARES: A Comprehensive Framework for Assessing Audio Encoder Performance","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:05:15.374000Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2505.16369"},"observation_digest":"sha256:618b9ec73aa7cd6c6604ae046903a2caf7bd6550a6399fac000686027c584595","observation_id":"f79a4b8a-b26f-42a1-b1b4-ab7a545d4d09","resolution":{"observed_at":"2026-08-07T15:05:15.374000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-07T14:48:06.850072Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Com- pact Models,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2505.17632","last_updated":"2025-05-23T08:45:46Z","snapshot_observed_at":"2026-08-07T14:41:08.766484Z","submitted_at":"2025-05-23T08:45:46Z","title":"ReqBrain: Task-Specific Instruction Tuning of LLMs for AI-Assisted Requirements Generation","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T14:48:06.850072Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2505.17632"},"observation_digest":"sha256:3b8213ac2d55840b59eb23dcab5acb23e20b8cb80f5506ae4e14c3e50d77da0c","observation_id":"1cf6706a-a972-4eb0-899f-44906e722334","resolution":{"observed_at":"2026-08-07T14:48:06.850072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-07T10:47:28.375567Z","title":"Iulia Turc, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova","venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2506.04566","last_updated":"2025-06-05T02:34:50Z","snapshot_observed_at":"2026-08-07T10:36:46.616273Z","submitted_at":"2025-06-05T02:34:50Z","title":"Clustering and Median Aggregation Improve Differentially Private Inference","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:47:28.375567Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2506.04566"},"observation_digest":"sha256:471a2ff499eb2280946db569cadb9c85bd63c2e4b2528dacadd84d2f93cff090","observation_id":"8a704e81-ed1e-4f6a-b3fc-de2c013d0472","resolution":{"observed_at":"2026-08-07T10:47:28.375567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-07T05:50:18.871863Z","title":"Well-read students learn better: On the importance of pre-training compact models","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.06926","last_updated":"2025-06-07T21:29:25Z","snapshot_observed_at":"2026-08-08T11:16:57.683716Z","submitted_at":"2025-06-07T21:29:25Z","title":"Basis Transformers for Multi-Task Tabular Regression","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T05:50:18.871863Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2506.06926"},"observation_digest":"sha256:8d5173242f4d47b4ef91a75b670cd3a6cad1faaa2d35c5d0c6367407329fd65b","observation_id":"45c3ebdf-44e2-4bb3-8b52-ab475807b07d","resolution":{"observed_at":"2026-08-07T05:50:18.871863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-06T23:57:26.565601Z","title":"arXiv preprint arXiv:1908.08962 (2019)","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.15681","last_updated":"2026-06-25T14:33:27Z","snapshot_observed_at":"2026-08-08T08:54:57.204006Z","submitted_at":"2025-06-18T17:59:49Z","title":"GenRecal: Generation after Recalibration from Large to Small Vision-Language Models","version":4},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-06T23:57:26.565601Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2506.15681"},"observation_digest":"sha256:6fe9d11ea0cf65734c3f216d0619965fd2db3451f4f1ab4fcda88e06a3fb66bd","observation_id":"b7083735-83d9-4ea7-b284-1bb1e179a483","resolution":{"observed_at":"2026-08-06T23:57:26.565601Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-06T20:36:45.789946Z","title":"Well-Read Students Learn Better: The Impact of Student Initialization on Knowledge Distillation,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2507.02364","last_updated":"2025-07-03T06:52:44Z","snapshot_observed_at":"2026-08-07T16:47:16.628644Z","submitted_at":"2025-07-03T06:52:44Z","title":"QFFN-BERT: An Empirical Study of Depth, Performance, and Data Efficiency in Hybrid Quantum-Classical Transformers","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:36:45.789946Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2507.02364"},"observation_digest":"sha256:37202dc0997f87601026eb1688c8504fe7384905e50e01ae9a0bfefd17d5c274","observation_id":"2275fc31-732d-48c8-8c54-46dc8e0b0da4","resolution":{"observed_at":"2026-08-06T20:36:45.789946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-06T18:49:30.548612Z","title":"arXiv preprint arXiv:1908.08962 (2019)","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.07280","last_updated":"2025-07-09T20:57:55Z","snapshot_observed_at":"2026-08-06T18:42:33.095139Z","submitted_at":"2025-07-09T20:57:55Z","title":"The Impact of Background Speech on Interruption Detection in Collaborative Groups","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T18:49:30.548612Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2507.07280"},"observation_digest":"sha256:0319add1bac86268ba3e1d8d27424c6d376ecbefa24fc732c350c1bf4258c705","observation_id":"36da8803-3ba9-4b50-ad02-c6a6d6676f75","resolution":{"observed_at":"2026-08-06T18:49:30.548612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-06T15:39:46.566529Z","title":"arXiv preprint arXiv:1908.08962","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2507.22918","last_updated":"2025-07-21T07:09:32Z","snapshot_observed_at":"2026-08-07T04:49:41.920479Z","submitted_at":"2025-07-21T07:09:32Z","title":"Semantic Convergence: Investigating Shared Representations Across Scaled LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T15:39:46.566529Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2507.22918"},"observation_digest":"sha256:72b967fe5c4cee24a117cd7bdf4e2549ae6490d0a3ec04abf144142be97a08bd","observation_id":"0200a562-5331-4b94-b02d-47c6bea20251","resolution":{"observed_at":"2026-08-06T15:39:46.566529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T19:00:30.108034Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.13533","last_updated":"2025-08-19T05:49:39Z","snapshot_observed_at":"2026-08-07T10:33:49.568008Z","submitted_at":"2025-08-19T05:49:39Z","title":"Compressed Models are NOT Trust-equivalent to Their Large Counterparts","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T19:00:30.108034Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2508.13533"},"observation_digest":"sha256:1151018a42b502a1bb5a0d7b33d2ea713d22f7332826df16d96e40aa1f68418b","observation_id":"6888f70e-d5f5-49e2-9b47-a3bf6254f2f2","resolution":{"observed_at":"2026-08-05T19:00:30.108034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-04T22:30:12.153879Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2509.07324","last_updated":"2025-09-09T01:43:48Z","snapshot_observed_at":"2026-08-06T16:39:00.860220Z","submitted_at":"2025-09-09T01:43:48Z","title":"Mitigating Attention Localization in Small Scale: Self-Attention Refinement via One-step Belief Propagation","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-04T22:30:12.153879Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2509.07324"},"observation_digest":"sha256:c284760273bbf6fbad7a844702641e4466fb47f6a721e413ef117c3e20571010","observation_id":"8642ce73-9232-43de-adca-746c0bc8e8b0","resolution":{"observed_at":"2026-08-04T22:30:12.153879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-03T06:36:47.013517Z","title":"Well-read students learn better: On the importance of pre-training compact models, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2601.22531","last_updated":"2026-05-27T20:55:02Z","snapshot_observed_at":"2026-08-07T23:30:48.750427Z","submitted_at":"2026-01-30T04:07:47Z","title":"Learn from A Rationalist: Distilling Intermediate Interpretable Rationales","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-03T06:36:47.013517Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2601.22531"},"observation_digest":"sha256:6bdc09150f68c9135351991add11eb0db026cee14d94e56124df943a22ae31b1","observation_id":"3b23346f-ddd7-47a2-838f-b8cc05b610ce","resolution":{"observed_at":"2026-08-03T06:36:47.013517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-02T23:43:34.290725Z","title":"Well-read students learn better: On the importance of pre-training compact models.arXiv preprint arXiv:1908.08962,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2602.12952","last_updated":"2026-07-30T10:36:55Z","snapshot_observed_at":"2026-08-02T23:54:38.696019Z","submitted_at":"2026-02-13T14:16:34Z","title":"Transporting Task Vectors across Different Architectures without Training","version":3},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-02T23:43:34.290725Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2602.12952"},"observation_digest":"sha256:15752a1699e1ab013e3cffd180487f768c7d11f64ecbce08f5edd7f29db6c4d3","observation_id":"2380fa43-c316-4fd3-95ac-7337dde7ea55","resolution":{"observed_at":"2026-08-02T23:43:34.290725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2603.20910","last_updated":"2026-04-04T14:02:00Z","snapshot_observed_at":"2026-08-06T13:43:37.795567Z","submitted_at":"2026-03-21T18:46:38Z","title":"LLM-ODE: Data-driven Discovery of Dynamical Systems with Large Language Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T06:30:29.297712Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2603.20910"},"observation_digest":"sha256:41fd3e046c5d62c5361c40228c8925c2c0cb6e572ff3fa9dbf5a6c60e14dd16b","observation_id":"b0f4d5e1-631f-4c05-a16f-886417ac6a32","resolution":{"observed_at":"2026-05-15T06:35:09.663608Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2605.07297","last_updated":"2026-05-08T06:08:25Z","snapshot_observed_at":"2026-08-06T12:28:28.001522Z","submitted_at":"2026-05-08T06:08:25Z","title":"Spectrum-Adaptive Generalization Bounds for Trained Deep Transformers","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-11T01:13:25.378895Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2605.07297"},"observation_digest":"sha256:134393197dc15d8fd048b6b16837843f92729eef89140e3999d774351068a5a7","observation_id":"53ce16ed-f4f4-4578-951e-6507980dc056","resolution":{"observed_at":"2026-05-11T04:35:58.818603Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2605.11695","last_updated":"2026-05-12T07:51:56Z","snapshot_observed_at":"2026-07-06T23:23:30.499023Z","submitted_at":"2026-05-12T07:51:56Z","title":"Emergent Communication between Heterogeneous Visual Agents through Decentralized Learning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T06:38:48.402207Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2605.11695"},"observation_digest":"sha256:d7f59db5f5c2fcaa5b8a8b8ce3c74030c968e1ed82a090860c6ad294a9705dd4","observation_id":"c11bb07a-c02e-4f2a-beba-32b1c119e058","resolution":{"observed_at":"2026-05-13T06:42:25.642869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2606.02100","last_updated":"2026-06-01T11:32:02Z","snapshot_observed_at":"2026-07-06T23:42:35.063977Z","submitted_at":"2026-06-01T11:32:02Z","title":"PortBERT: Navigating the Depths of Portuguese Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-28T14:43:28.401687Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2606.02100"},"observation_digest":"sha256:1641bee9538d65d33786531404d5952bf9c636017e488bc61371c03d8870088f","observation_id":"a9ea690a-c066-4608-8fe2-7cce1f3c4b50","resolution":{"observed_at":"2026-07-01T23:06:20.160364Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2606.21240","last_updated":"2026-06-19T09:08:43Z","snapshot_observed_at":"2026-08-08T10:28:57.871957Z","submitted_at":"2026-06-19T09:08:43Z","title":"DIPBox: A Multi-scale Testing Framework for Tracking Dataset Regeneration","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-26T14:15:19.553221Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2606.21240"},"observation_digest":"sha256:1ef9193065b336f08fce0736f68aea74c437e7e1fcdf567a0b80f2672b82774d","observation_id":"b12765b1-cbca-4ba6-a02b-d1d5ba1b9dcc","resolution":{"observed_at":"2026-07-04T06:39:38.256949Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2606.21593","last_updated":"2026-06-25T13:10:27Z","snapshot_observed_at":"2026-08-08T11:31:21.836492Z","submitted_at":"2026-06-19T16:49:50Z","title":"Geometric and Information Compression of Representations in Deep Learning","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-26T14:51:57.746071Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2606.21593"},"observation_digest":"sha256:6f65dac61f9408b768a18467866bbea1bf984ff8408262de3a20502d83e7ca56","observation_id":"ca8b2d3f-9ec1-4b31-b6cc-096821f93aea","resolution":{"observed_at":"2026-07-04T06:09:36.862161Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":"1908.08962","doi":"10.48550/arxiv.1908.08962","metadata_source":"arxiv_reference","pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Well-read students learn better: On the impor- tance of pre-training compact models.arXiv preprint arXiv:1908.08962","venue":"arXiv (Cornell University)","work_id":"73c687fa-7f5c-4e66-9c65-4a30cd2d5c33","year":1908},"citing_paper":{"arxiv_id":"2606.22327","last_updated":"2026-06-24T14:02:53Z","snapshot_observed_at":"2026-08-07T17:40:04.855669Z","submitted_at":"2026-06-21T04:05:38Z","title":"Geometry-Aware Online Scheduling for LLM Serving: From Theoretical Bound to System Practice","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-26T11:17:35.860286Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2606.22327"},"observation_digest":"sha256:a53220f4e639dba30bf6b034d20cb7f6f9e55b0478215c52bf745d0235b61d95","observation_id":"8f78a028-d1f0-41c2-9611-f9ce741e7327","resolution":{"observed_at":"2026-07-04T08:39:41.972877Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.08962","snapshot_observed_at":"2026-08-01T18:14:43.084276Z","title":null,"venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2607.17382","last_updated":"2026-07-19T18:48:59Z","snapshot_observed_at":"2026-08-07T10:11:51.522478Z","submitted_at":"2026-07-19T18:48:59Z","title":"Team DACTYL at PAN 2026: Bayesian Data Mixing and Empirical X-risk Minimization for AI-text Detection","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-01T18:14:43.084276Z"},"links":{"cited_paper":"/paper/1908.08962","citing_paper":"/paper/2607.17382"},"observation_digest":"sha256:343854991ab1264f224635ce63bc6c0505f539504d5d88c186489a66acaa2886","observation_id":"0c0743ae-b6b1-426a-9ec2-4a42530d799f","resolution":{"observed_at":"2026-08-01T18:14:43.084276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1908.08962/citation-record","integrity":"/paper/1908.08962/integrity","json":"/paper/1908.08962/citation-record.json","paper":"/paper/1908.08962"},"outbound":[],"paper":{"arxiv_id":"1908.08962","last_updated":"2019-09-25T22:55:20Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-01T21:55:51.630328Z","submitted_at":"2019-08-23T18:02:05Z","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 25 inbound Pith citation observations for arXiv:1908.08962."}