{"as_of":"2026-08-07T22:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:983b9774b3c9d3366b523fd4e547e7226b9d759322a06d552805b38723cab913","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:20:01.458159Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-29T05:14:12.795412Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T08:59:42.571077Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"cited_work":{"arxiv_id":"2506.05695","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05695","snapshot_observed_at":"2026-07-04T08:59:42.571077Z","title":"Being strong progressively! enhancing knowledge distillation of large language models through a curriculum learning framework","venue":null,"work_id":"50166d6e-b145-4829-8c22-1a0062db5e8f","year":2025},"citing_paper":{"arxiv_id":"2605.11260","last_updated":"2026-05-11T21:37:20Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T21:37:20Z","title":"Curriculum Learning-Guided Progressive Distillation in Large Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T01:53:21.920498Z"},"links":{"cited_paper":"/paper/2506.05695","citing_paper":"/paper/2605.11260"},"observation_digest":"sha256:1c09f15b67feb506eb52ca1fcd7d0524355af0e634f54be9275fd43c1028b94a","observation_id":"eda96db4-1ee8-4b8f-9031-796e77a95c74","resolution":{"observed_at":"2026-05-13T01:57:05.839700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"cited_work":{"arxiv_id":"2506.05695","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05695","snapshot_observed_at":"2026-07-04T08:59:42.571077Z","title":"Being strong progressively! enhancing knowledge distillation of large language models through a curriculum learning framework","venue":null,"work_id":"50166d6e-b145-4829-8c22-1a0062db5e8f","year":2025},"citing_paper":{"arxiv_id":"2606.22600","last_updated":"2026-06-26T02:45:32Z","snapshot_observed_at":"2026-08-05T19:54:05.746402Z","submitted_at":"2026-06-21T17:20:21Z","title":"On the Position Bias of On-Policy Distillation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-26T10:45:16.028763Z"},"links":{"cited_paper":"/paper/2506.05695","citing_paper":"/paper/2606.22600"},"observation_digest":"sha256:4c97f86ad311a2d081868d2ece1c2cac3704989de5e0a5a9c3f0012ed03293f1","observation_id":"a08082e3-8364-4216-b925-ec25bacfd720","resolution":{"observed_at":"2026-07-04T08:59:42.572759Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"cited_work":{"arxiv_id":"2506.05695","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05695","snapshot_observed_at":"2026-07-04T08:59:42.571077Z","title":"Being strong progressively! enhancing knowledge distillation of large language models through a curriculum learning framework","venue":null,"work_id":"50166d6e-b145-4829-8c22-1a0062db5e8f","year":2025},"citing_paper":{"arxiv_id":"2606.22600","last_updated":"2026-06-26T02:45:32Z","snapshot_observed_at":"2026-08-05T19:54:05.746402Z","submitted_at":"2026-06-21T17:20:21Z","title":"On the Position Bias of On-Policy Distillation","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-29T05:14:12.795412Z"},"links":{"cited_paper":"/paper/2506.05695","citing_paper":"/paper/2606.22600"},"observation_digest":"sha256:ed973a3b7131785099b92232dab68eaeda51df9e514c90e577617b112c98359d","observation_id":"6226bddf-3b5e-4b17-88d2-2e8561ce7600","resolution":{"observed_at":"2026-06-29T18:13:49.263364Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.05695/citation-record","integrity":"/paper/2506.05695/integrity","json":"/paper/2506.05695/citation-record.json","paper":"/paper/2506.05695"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-07T10:20:01.279316Z","title":"Distilling the knowledge in a neural network","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.279316Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:d6e4ef7195c09b8f14d6f8cc6adc633081ec5378ff1bdb6deaa0ad528b8bad19","observation_id":"cfe8611f-0d0e-4692-8726-195db827e506","resolution":{"observed_at":"2026-08-07T10:20:01.279316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15190","last_updated":"2023-07-27T20:39:06Z","snapshot_observed_at":"2026-07-06T15:59:30.956483Z","submitted_at":"2023-07-27T20:39:06Z","title":"f-Divergence Minimization for Sequence-Level Knowledge Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15190","snapshot_observed_at":"2026-08-07T10:20:01.284265Z","title":"F-divergence minimization for sequence-level knowledge distillation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.284265Z"},"links":{"cited_paper":"/paper/2307.15190","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:e80046652e324e247e077d555191565a6635cea0e06017325c38726e8592ef09","observation_id":"99bdbf87-0aff-4dcd-b962-80c1bea9f7ac","resolution":{"observed_at":"2026-08-07T10:20:01.284265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03898","last_updated":"2024-07-03T04:57:41Z","snapshot_observed_at":"2026-08-05T01:48:55.895470Z","submitted_at":"2024-02-06T11:10:35Z","title":"DistiLLM: Towards Streamlined Distillation for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03898","snapshot_observed_at":"2026-08-07T10:20:01.289189Z","title":"Distillm: Towards streamlined distillation for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.289189Z"},"links":{"cited_paper":"/paper/2402.03898","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:da6635d38d8374d5b72ae269d15e54e88c0db7cefe2fe6fa374b381d5e4ad5d9","observation_id":"087c71f0-6667-4297-8cc2-07fcdc99ece5","resolution":{"observed_at":"2026-08-07T10:20:01.289189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.294164Z","title":"On-policy distillation of language models: Learning from self-generated mistakes","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.294164Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:3e7d5d32cc60dea3b5e59afdf6e440d532a59e15bfb8e09beb704399198c3da6","observation_id":"ec85ab1b-c6d9-4403-bfdb-825bf0e744d2","resolution":{"observed_at":"2026-08-07T10:20:01.294164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.298578Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.298578Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:87af691ccbbe564af37fe7cebd6bef83854cc62e532d7f971d54a82cbc13cf1d","observation_id":"d449558b-00b9-4508-a59c-fed70d04db23","resolution":{"observed_at":"2026-08-07T10:20:01.298578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.08061","last_updated":"2023-08-15T22:26:58Z","snapshot_observed_at":"2026-07-06T16:06:38.424057Z","submitted_at":"2023-08-15T22:26:58Z","title":"The Costly Dilemma: Generalization, Evaluation and Cost-Optimal Deployment of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.08061","snapshot_observed_at":"2026-08-07T10:20:01.302831Z","title":"The costly dilemma: generalization, evaluation and cost-optimal deployment of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.302831Z"},"links":{"cited_paper":"/paper/2308.08061","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:db59a65375a649b5e16e4eb5fdb3b7cf54ba01a446c9ba76655df35befb82662","observation_id":"ff6082c2-8adb-4f8c-b719-98273d3ab173","resolution":{"observed_at":"2026-08-07T10:20:01.302831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:02.012399Z","title":"Pre-trained language models for text generation: A survey","venue":null,"work_id":"4d02d702-27a9-49a9-877b-a0cd2f1f176e","year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.308114Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:89d845c527f6da3674edc27ac6e6f9613eade22bd5c1574ee3549efd9839e3ab","observation_id":"76182876-9682-4f62-a43f-5ab776362382","resolution":{"observed_at":"2026-08-07T10:20:02.016507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.999772Z","title":"Confucius: Iterative tool learning from introspection feedback by easy-to-difficult curriculum","venue":null,"work_id":"62e8304e-e8eb-4a89-87d3-7ce5a3f5112a","year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.312564Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:f1aca66495d141cdf94547fdf7e7263a3362aa452f5d1a673d19b45198eb020e","observation_id":"554cb334-70ec-4744-890d-4030551af22a","resolution":{"observed_at":"2026-08-07T10:20:02.003839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.986724Z","title":"Llama 3.2: Revolutionizing edge ai and vision with open, customizable models","venue":null,"work_id":"b337d69f-80af-44c6-800e-bdcf4486b2a1","year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.316246Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:f2c6bc3d9f8f8bb79084cb3e1cab72736f98e6678b3be9e088c17010bda79777","observation_id":"b2aae059-d1b1-4590-82c9-8ef2f82a2cce","resolution":{"observed_at":"2026-08-07T10:20:01.990963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T10:20:01.320067Z","title":"Gemma 2: Improving open language models at a practical size","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.320067Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:9c5c0d5673f97554b8e9cb7bf48f7529d7ce9f69826fceac8cc4771ec989dcf8","observation_id":"7145e623-0141-4c7a-8ec7-247a8d2b7057","resolution":{"observed_at":"2026-08-07T10:20:01.320067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.324172Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.324172Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:37bc0559907c1bfc246dd91ebe670a252be7589b096d4a0a83646ac30107d18e","observation_id":"909d9c47-2dd2-4c7c-9acf-444bcda42be9","resolution":{"observed_at":"2026-08-07T10:20:01.324172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13116","last_updated":"2024-10-21T16:22:33Z","snapshot_observed_at":"2026-08-02T17:17:56.296462Z","submitted_at":"2024-02-20T16:17:37Z","title":"A Survey on Knowledge Distillation of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13116","snapshot_observed_at":"2026-08-07T10:20:01.327884Z","title":"A survey on knowledge distillation of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.327884Z"},"links":{"cited_paper":"/paper/2402.13116","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:99b0bad8cfda713b3ed3d4f5461c745c3ecff991f6813e5e08f0f627f3901418","observation_id":"9ad5a208-92f2-4fbd-b237-142226846866","resolution":{"observed_at":"2026-08-07T10:20:01.327884Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.962218Z","title":"Survey on knowledge distillation for large language models: methods, evaluation, and application","venue":null,"work_id":"478eae27-3dae-4758-a7e7-3392acb0ef6c","year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.331845Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:f8855f418a8a7cd07fd46bd121869adf7fd7bc52891881defaad808dcb28870b","observation_id":"8486cf3f-9043-4f36-835c-e3c1c610d311","resolution":{"observed_at":"2026-08-07T10:20:01.967781Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.948539Z","title":"Sequence-level knowledge distillation","venue":null,"work_id":"2c751771-bc68-43b2-a48c-ccff9516ab0b","year":2016},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.335570Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:ad7af402fc55fa98a81652c32ec5ffcbbc31c78a8ad2b0f4c4bdb7d6a5dfb3d8","observation_id":"3fef852f-34a6-4a9a-a561-875ad5606993","resolution":{"observed_at":"2026-08-07T10:20:01.952778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T10:20:01.339406Z","title":"Gpt-4o system card","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.339406Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:2bf12a1b7350e993a2a0f597119cbf01c78413dd6e9aec442f21ed26296f16cb","observation_id":"e91cefaa-b4c5-4dd6-95d6-cf4dca159b0f","resolution":{"observed_at":"2026-08-07T10:20:01.339406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.935820Z","title":"Claude 3.5 Sonnet","venue":null,"work_id":"0a4a9220-673f-42c8-b946-6dc3ce6f04b3","year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.343014Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:113be075e53b7bee0ee2a2a6da85dbc667492be4b88144fe4b373afa5b61076c","observation_id":"dfdb7dd3-5fdc-4aa1-ac08-63e274d34596","resolution":{"observed_at":"2026-08-07T10:20:01.940346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T10:20:01.346911Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.346911Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:1381c59b3fc426e2b6414e33196a0d99c4d31e89611613f731d561663baff305","observation_id":"bf021401-785a-4cd0-9c92-7afa6b871046","resolution":{"observed_at":"2026-08-07T10:20:01.346911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.07164","last_updated":"2024-01-26T07:04:02Z","snapshot_observed_at":"2026-07-06T15:53:53.845428Z","submitted_at":"2023-07-14T05:23:08Z","title":"Learning to Retrieve In-Context Examples for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.07164","snapshot_observed_at":"2026-08-07T10:20:01.350926Z","title":"Learning to retrieve in-context examples for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.350926Z"},"links":{"cited_paper":"/paper/2307.07164","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:e32bddff11a6d52146e1b13427c91800bb50008342d365d63c13003de584c9b2","observation_id":"501a2670-797f-4e85-91e8-5a146080751f","resolution":{"observed_at":"2026-08-07T10:20:01.350926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.02301","last_updated":"2023-07-05T16:59:31Z","snapshot_observed_at":"2026-07-06T15:22:55.122322Z","submitted_at":"2023-05-03T17:50:56Z","title":"Distilling Step-by-Step! Outperforming Larger Language Models with Less Training Data and Smaller Model Sizes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.02301","snapshot_observed_at":"2026-08-07T10:20:01.355334Z","title":"Distilling step-by-step! outperforming larger language models with less training data and smaller model sizes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.355334Z"},"links":{"cited_paper":"/paper/2305.02301","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:2d7b79337811302bacfee31b4cd145e696416eb76d52989a585fc83a73d031b3","observation_id":"a8f379ae-1790-4372-96fd-6daaf0c6884b","resolution":{"observed_at":"2026-08-07T10:20:01.355334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-07T10:20:01.359365Z","title":"Instruction tuning with gpt-4","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.359365Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:8c3b7b8fb01a70c5615a9648b83788d4e16156bacb3a0ee501d0639cd2589338","observation_id":"4e6b7ab7-546e-4c46-9f10-18aa6b51ec3e","resolution":{"observed_at":"2026-08-07T10:20:01.359365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-07T10:20:01.363233Z","title":"Deepseek-v3 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.363233Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:11ac394878774e33ab42ac9a4f12098ae5b2048e969293c15f5f570c0f9c5670","observation_id":"22dd1424-f88b-40b2-8377-9f339a836db4","resolution":{"observed_at":"2026-08-07T10:20:01.363233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T10:20:01.367597Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.367597Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:421be70a93fa0bd90c447458fae1ba499e277f4b117b3802e600c71906e120c6","observation_id":"2531d07b-e529-44a1-8225-c0ac21d390d5","resolution":{"observed_at":"2026-08-07T10:20:01.367597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.923307Z","title":"The ai index 2025 annual report","venue":null,"work_id":"43aa29e9-1824-4246-99e4-f325a420d469","year":2025},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.371223Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:059f94af10b9d032e48a01e1da67c828eafc29510d4fd57293b7f5d9a10867b1","observation_id":"4c20a7dc-d7bd-45c2-b779-8600b3dadb3c","resolution":{"observed_at":"2026-08-07T10:20:01.927262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.08543","last_updated":"2026-01-31T09:16:35Z","snapshot_observed_at":"2026-08-02T20:03:32.798288Z","submitted_at":"2023-06-14T14:44:03Z","title":"MiniLLM: On-Policy Distillation of Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.08543","snapshot_observed_at":"2026-08-07T10:20:01.375136Z","title":"Minillm: Knowledge distillation of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.375136Z"},"links":{"cited_paper":"/paper/2306.08543","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:fbae882bd3a6a0d6ee22189645691daaf28410ec50d513c230d84c53f17e928b","observation_id":"13e65741-fe89-4980-a479-afc08823d841","resolution":{"observed_at":"2026-08-07T10:20:01.375136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.07253","last_updated":"2020-10-29T00:40:45Z","snapshot_observed_at":"2026-08-03T18:52:39.630483Z","submitted_at":"2020-09-15T17:43:02Z","title":"Autoregressive Knowledge Distillation through Imitation Learning","version":2},"cited_work":{"arxiv_id":"2009.07253","doi":null,"metadata_source":"pith","pith_arxiv_id":"2009.07253","snapshot_observed_at":"2026-08-07T10:20:01.598781Z","title":"Autoregressive Knowledge Distillation through Imitation Learning","venue":"cs.CL","work_id":"c096915c-1b74-48a8-9ea2-9b643752d37a","year":2020},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.379230Z"},"links":{"cited_paper":"/paper/2009.07253","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:77e053d179ef38c571002ee6e8304a15f0c1bd967f1ff6ff3b13e97fc7da273c","observation_id":"70bf6e29-f7cc-4636-ac7f-62e44b09a863","resolution":{"observed_at":"2026-08-07T10:20:01.605055Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11325","last_updated":"2025-04-27T23:18:29Z","snapshot_observed_at":"2026-07-06T19:33:39.134036Z","submitted_at":"2024-10-15T06:51:25Z","title":"Speculative Knowledge Distillation: Bridging the Teacher-Student Gap Through Interleaved Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11325","snapshot_observed_at":"2026-08-07T10:20:01.383167Z","title":"Speculative knowledge distillation: Bridging the teacher-student gap through interleaved sampling","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.383167Z"},"links":{"cited_paper":"/paper/2410.11325","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:5690efd47c16c02b72a70e34b32976e606b1af2d68980a7f22f57f543ff6f1dd","observation_id":"a41afb16-ac05-4c5b-85d2-c0662c65a0e7","resolution":{"observed_at":"2026-08-07T10:20:01.383167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.910963Z","title":"A survey on curriculum learning","venue":null,"work_id":"1c765de2-6d76-4dd9-bbe8-b7bbde566059","year":2021},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.387674Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:b8a67d622a4a16b478d8e174759ec009465d1504fe60abfaddf31fcb00cdaf5e","observation_id":"eea00772-dfae-42b5-b0b4-7ecc4f2918d8","resolution":{"observed_at":"2026-08-07T10:20:01.914929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.897347Z","title":"Science and practice of strength training","venue":null,"work_id":"1df4ebde-83e6-48f3-a978-cae8031b1e25","year":2020},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.391904Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:6fe80cac28560e9ebe670eb60d318a59709ab5871721b5a9c9e13c9e2448760b","observation_id":"ee208c69-13c5-4d72-8f42-c13b2aa9dc20","resolution":{"observed_at":"2026-08-07T10:20:01.901922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17328","last_updated":"2024-10-01T16:45:12Z","snapshot_observed_at":"2026-08-03T10:31:12.322733Z","submitted_at":"2024-06-25T07:25:15Z","title":"Dual-Space Knowledge Distillation for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17328","snapshot_observed_at":"2026-08-07T10:20:01.395817Z","title":"Dual-space knowledge distillation for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.395817Z"},"links":{"cited_paper":"/paper/2406.17328","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:ba2c526d58768e0df1727a5fe679f94a117b0feb0c06d475ca4c67592e1804ea","observation_id":"3787f6e1-e927-498d-817e-42da0b49e7eb","resolution":{"observed_at":"2026-08-07T10:20:01.395817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04836","last_updated":"2024-06-07T11:09:13Z","snapshot_observed_at":"2026-07-06T18:27:03.044022Z","submitted_at":"2024-06-07T11:09:13Z","title":"Revisiting Catastrophic Forgetting in Large Language Model Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04836","snapshot_observed_at":"2026-08-07T10:20:01.399991Z","title":"Revisiting catastrophic forgetting in large language model tuning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.399991Z"},"links":{"cited_paper":"/paper/2406.04836","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:263d5c9a82052ff3de60abc8bf5eeeff2980ce8161ab705dc9341697fdf4017b","observation_id":"f38d6d3c-12ca-44f1-9aa6-febd0d802f9c","resolution":{"observed_at":"2026-08-07T10:20:01.399991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.403847Z","title":"Rouge: A package for automatic evaluation of summaries","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.403847Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:64ec80e5d55f8a11927f5a95754e680d8c16ebd1c53475ce65bcf781db8db97d","observation_id":"d62ba192-4517-4bbb-b323-60ac91ea3a8f","resolution":{"observed_at":"2026-08-07T10:20:01.403847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.875413Z","title":"Reciprocal rank fusion outperforms condorcet and individual rank learning methods","venue":null,"work_id":"de4d2c32-838f-4b87-ad65-9cd1f73cac02","year":2009},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.407506Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:84597f948dd169816b646704fd07dd8785a8e2856d7bea7474b38e00d7c46712","observation_id":"48038b3f-e65f-4447-b8f2-40563458cd2d","resolution":{"observed_at":"2026-08-07T10:20:01.880064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.411150Z","title":"Curriculum learning","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.411150Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:f744ed2ad6822a24e5e40fbe0e16d74915f84a485caf8c371e9298fe861e4048","observation_id":"941cc7dd-cb52-4f21-9f7d-0a80f01bb12e","resolution":{"observed_at":"2026-08-07T10:20:01.411150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.854291Z","title":"A comparison of most-to-least and least-to-most prompting on the acquisition of solitary play skills","venue":null,"work_id":"c28e0c5b-fea2-4de5-a619-f79672d38458","year":2008},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.414884Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:7e6a2cf8040da3c6abf4c874dae151aafdf2e6d3544997840c79f2031bb67a8b","observation_id":"346ecbfb-d1e5-45da-9445-991f73847409","resolution":{"observed_at":"2026-08-07T10:20:01.858626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.418825Z","title":"Free dolly: Introducing the world’s first truly open instruction-tuned llm, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.418825Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:d267fa7db19eae48191e66b1689f7dafa27f6f51bc61ce240d2c3f4a1bbe0e94","observation_id":"ec28595b-bcb1-4080-9b71-d5b9e3757c26","resolution":{"observed_at":"2026-08-07T10:20:01.418825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-07-06T14:33:11.945106Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-08-07T10:20:01.422462Z","title":"Self-instruct: Aligning language models with self-generated instructions.arXiv preprint arXiv:2212.10560, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.422462Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:5cfb0f190293a53da185ef14415973ce21adcf587fa1fda95ccf57d32f8b99b8","observation_id":"98f27c66-8594-4e69-b199-c6505d9e1f58","resolution":{"observed_at":"2026-08-07T10:20:01.422462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.426657Z","title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.426657Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:66ac6ebfb89da12ac78b66f549e649a246f2b86979715eac85baa7ab19b2e58f","observation_id":"45a81177-c92b-4e73-8837-03a68102db57","resolution":{"observed_at":"2026-08-07T10:20:01.426657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.07705","last_updated":"2022-10-24T07:00:15Z","snapshot_observed_at":"2026-07-06T13:00:53.618234Z","submitted_at":"2022-04-16T03:12:30Z","title":"Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ NLP Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.07705","snapshot_observed_at":"2026-08-07T10:20:01.430323Z","title":"Benchmarking generalization via in-context instructions on 1,600+ language tasks","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.430323Z"},"links":{"cited_paper":"/paper/2204.07705","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:742fc511ca255ae2f0ed8f11e454ba7d50dbd01eaf4822a5e311688499fb5eeb","observation_id":"47cfcfbe-e28c-45e5-a7f0-6c88f257d8a4","resolution":{"observed_at":"2026-08-07T10:20:01.430323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.09689","last_updated":"2022-12-19T18:21:00Z","snapshot_observed_at":"2026-07-06T14:32:33.541029Z","submitted_at":"2022-12-19T18:21:00Z","title":"Unnatural Instructions: Tuning Language Models with (Almost) No Human Labor","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.09689","snapshot_observed_at":"2026-08-07T10:20:01.434515Z","title":"Unnatural instructions: Tuning language models with (almost) no human labor","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.434515Z"},"links":{"cited_paper":"/paper/2212.09689","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:ad242f123d265a2f34e7e8e09148dcd94b76bdda26520b4d32149c46af7dbce9","observation_id":"893df899-ab89-4c4f-9307-5cdd9138c26e","resolution":{"observed_at":"2026-08-07T10:20:01.434515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.824456Z","title":"Learning from multiple experts: Self-paced knowledge distillation for long-tailed classification","venue":null,"work_id":"cf9e77d1-5e93-4c4c-8793-f30ec0edaf38","year":2020},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.438405Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:9cbca462cd3d7ff488e9e96a7356ff28253fc13f7ea25667a40d0aa0ff170c33","observation_id":"df4d3779-68c0-4e42-b580-b3b4a531eede","resolution":{"observed_at":"2026-08-07T10:20:01.829077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.809948Z","title":"Curriculum tem- perature for knowledge distillation","venue":null,"work_id":"abd62a44-eb3f-4902-9b02-269568035a4a","year":2023},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.442260Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:66601279b884976354a50497e03ab4f22edd8f473aa07c9ad81d5e598a7ea5e7","observation_id":"eab608f7-cb61-43c2-9c11-ca6d61773501","resolution":{"observed_at":"2026-08-07T10:20:01.814690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07067","last_updated":"2025-05-30T04:41:12Z","snapshot_observed_at":"2026-08-07T17:17:34.718676Z","submitted_at":"2025-03-10T08:51:32Z","title":"DistiLLM-2: A Contrastive Approach Boosts the Distillation of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.07067","snapshot_observed_at":"2026-08-07T10:20:01.446134Z","title":"Distillm-2: A contrastive approach boosts the distillation of llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.446134Z"},"links":{"cited_paper":"/paper/2503.07067","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:e1a9aacfda5113eb75121dcc83ce3d4b67f892ff0c05539d01cd4a329142c32c","observation_id":"144e9747-c2a9-48fd-ba9d-77015d787b9e","resolution":{"observed_at":"2026-08-07T10:20:01.446134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.450127Z","title":"Language models are unsupervised multitask learners","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.450127Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:61bf0e3f355c59f63f6372cc11c9e1165bf25f3503e6606fa7061d65a46f0874","observation_id":"26164519-8375-4b24-8118-c2914c9f93b9","resolution":{"observed_at":"2026-08-07T10:20:01.450127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-07T10:20:01.454112Z","title":"Opt: Open pre-trained transformer language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.454112Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:2f4ce7849073db902dc9c0cc991c81d477cff479c76decda7b9384c46e1d8c84","observation_id":"cb378345-090f-45fb-952d-19319c059348","resolution":{"observed_at":"2026-08-07T10:20:01.454112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:20:01.789122Z","title":"Qwen3: The latest large language model series from alibaba cloud, 2025","venue":null,"work_id":"f1b4c985-31fb-4bac-b792-94eacdac3749","year":2025},"citing_paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T10:20:01.458159Z"},"links":{"citing_paper":"/paper/2506.05695"},"observation_digest":"sha256:cb3ad352d8469e720c059593e871b2bbd3b6cad4aa547c9222693f158e1d84a8","observation_id":"d907fa2b-8ea7-4e2c-ba86-16dad332578e","resolution":{"observed_at":"2026-08-07T10:20:01.793316Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.05695","last_updated":"2025-06-06T02:48:38Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T12:30:36.888311Z","submitted_at":"2025-06-06T02:48:38Z","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":1,"verified_fuzzy":13},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 3 inbound Pith citation observations for arXiv:2506.05695."}