{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:KGFOB2CSGPBL55QJ2E6TX2ZG6Z","short_pith_number":"pith:KGFOB2CS","schema_version":"1.0","canonical_sha256":"518ae0e85233c2bef609d13d3beb26f65b5383824268fa49d64d4ac0288a661f","source":{"kind":"arxiv","id":"2109.11295","version":1},"attestation_state":"computed","paper":{"title":"Dynamic Knowledge Distillation for Pre-trained Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jie Zhou, Lei Li, Peng Li, Shuhuai Ren, Xu Sun, Yankai Lin","submitted_at":"2021-09-23T11:02:24Z","abstract_excerpt":"Knowledge distillation~(KD) has been proved effective for compressing large-scale pre-trained language models. However, existing methods conduct KD statically, e.g., the student model aligns its output distribution to that of a selected teacher model on the pre-defined training dataset. In this paper, we explore whether a dynamic knowledge distillation that empowers the student to adjust the learning procedure according to its competency, regarding the student performance and learning efficiency. We explore the dynamical adjustments on three aspects: teacher model adoption, data selection, and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.11295","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-09-23T11:02:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"23a17de42d5cfecc41eb87092990a035555ad3cb0cd94658cc2683b204da3b51","abstract_canon_sha256":"c2a4cf93c3c7730f08d21613005579ce2ec93ee7247c0aa3358255bf76ed4dfe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:16:56.514454Z","signature_b64":"q4ttzk49MS0EcafObns5StKOItrFi0tFJtEbrvK9Zaf7Z7d1qho34rsCzJkl2umfK4XOnAgxzoMrQGWpZYslBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"518ae0e85233c2bef609d13d3beb26f65b5383824268fa49d64d4ac0288a661f","last_reissued_at":"2026-07-05T03:16:56.514023Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:16:56.514023Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamic Knowledge Distillation for Pre-trained Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jie Zhou, Lei Li, Peng Li, Shuhuai Ren, Xu Sun, Yankai Lin","submitted_at":"2021-09-23T11:02:24Z","abstract_excerpt":"Knowledge distillation~(KD) has been proved effective for compressing large-scale pre-trained language models. However, existing methods conduct KD statically, e.g., the student model aligns its output distribution to that of a selected teacher model on the pre-defined training dataset. In this paper, we explore whether a dynamic knowledge distillation that empowers the student to adjust the learning procedure according to its competency, regarding the student performance and learning efficiency. We explore the dynamical adjustments on three aspects: teacher model adoption, data selection, and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.11295","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.11295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.11295","created_at":"2026-07-05T03:16:56.514087+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.11295v1","created_at":"2026-07-05T03:16:56.514087+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.11295","created_at":"2026-07-05T03:16:56.514087+00:00"},{"alias_kind":"pith_short_12","alias_value":"KGFOB2CSGPBL","created_at":"2026-07-05T03:16:56.514087+00:00"},{"alias_kind":"pith_short_16","alias_value":"KGFOB2CSGPBL55QJ","created_at":"2026-07-05T03:16:56.514087+00:00"},{"alias_kind":"pith_short_8","alias_value":"KGFOB2CS","created_at":"2026-07-05T03:16:56.514087+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.01649","citing_title":"Distilled Pretraining: A modern lens of Data, In-Context Learning and Test-Time Scaling","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z","json":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z.json","graph_json":"https://pith.science/api/pith-number/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/graph.json","events_json":"https://pith.science/api/pith-number/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/events.json","paper":"https://pith.science/paper/KGFOB2CS"},"agent_actions":{"view_html":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z","download_json":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z.json","view_paper":"https://pith.science/paper/KGFOB2CS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.11295&json=true","fetch_graph":"https://pith.science/api/pith-number/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/graph.json","fetch_events":"https://pith.science/api/pith-number/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/action/storage_attestation","attest_author":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/action/author_attestation","sign_citation":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/action/citation_signature","submit_replication":"https://pith.science/pith/KGFOB2CSGPBL55QJ2E6TX2ZG6Z/action/replication_record"}},"created_at":"2026-07-05T03:16:56.514087+00:00","updated_at":"2026-07-05T03:16:56.514087+00:00"}