{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:PX5KFZ3QOW7ATUGBXTOQYBODOL","short_pith_number":"pith:PX5KFZ3Q","schema_version":"1.0","canonical_sha256":"7dfaa2e77075be09d0c1bcdd0c05c372fc30a8b9166aa9b565fc03a72797dd03","source":{"kind":"arxiv","id":"2106.04570","version":3},"attestation_state":"computed","paper":{"title":"BERT Learns to Teach: Knowledge Distillation with Meta Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Canwen Xu, Julian McAuley, Wangchunshu Zhou","submitted_at":"2021-06-08T17:59:03Z","abstract_excerpt":"We present Knowledge Distillation with Meta Learning (MetaDistil), a simple yet effective alternative to traditional knowledge distillation (KD) methods where the teacher model is fixed during training. We show the teacher network can learn to better transfer knowledge to the student network (i.e., learning to teach) with the feedback from the performance of the distilled student network in a meta learning framework. Moreover, we introduce a pilot update mechanism to improve the alignment between the inner-learner and meta-learner in meta learning algorithms that focus on an improved inner-lea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.04570","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-08T17:59:03Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"ac5b9321123331dde7bda79918db965d14fd16289c121bbb489424600bf819e3","abstract_canon_sha256":"a24eaf81ca6cfa0b3c4cb24a7d49c17a599305944f552cb7f5c0b455e6edc43e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:50.613186Z","signature_b64":"R2gJSw/QlzxU5VfjTt2u0uZCkub7pCsNa2jiZRvODMJawOkgcW81Jq3NYkCsdG+6rbaw+6PMxPJ/jAUt6GY3AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7dfaa2e77075be09d0c1bcdd0c05c372fc30a8b9166aa9b565fc03a72797dd03","last_reissued_at":"2026-07-05T04:10:50.612720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:50.612720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BERT Learns to Teach: Knowledge Distillation with Meta Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Canwen Xu, Julian McAuley, Wangchunshu Zhou","submitted_at":"2021-06-08T17:59:03Z","abstract_excerpt":"We present Knowledge Distillation with Meta Learning (MetaDistil), a simple yet effective alternative to traditional knowledge distillation (KD) methods where the teacher model is fixed during training. We show the teacher network can learn to better transfer knowledge to the student network (i.e., learning to teach) with the feedback from the performance of the distilled student network in a meta learning framework. Moreover, we introduce a pilot update mechanism to improve the alignment between the inner-learner and meta-learner in meta learning algorithms that focus on an improved inner-lea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.04570","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.04570/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.04570","created_at":"2026-07-05T04:10:50.612770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.04570v3","created_at":"2026-07-05T04:10:50.612770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.04570","created_at":"2026-07-05T04:10:50.612770+00:00"},{"alias_kind":"pith_short_12","alias_value":"PX5KFZ3QOW7A","created_at":"2026-07-05T04:10:50.612770+00:00"},{"alias_kind":"pith_short_16","alias_value":"PX5KFZ3QOW7ATUGB","created_at":"2026-07-05T04:10:50.612770+00:00"},{"alias_kind":"pith_short_8","alias_value":"PX5KFZ3Q","created_at":"2026-07-05T04:10:50.612770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.04636","citing_title":"Put Teacher in Student's Shoes: Cross-Distillation for Ultra-compact Model Compression Framework","ref_index":54,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL","json":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL.json","graph_json":"https://pith.science/api/pith-number/PX5KFZ3QOW7ATUGBXTOQYBODOL/graph.json","events_json":"https://pith.science/api/pith-number/PX5KFZ3QOW7ATUGBXTOQYBODOL/events.json","paper":"https://pith.science/paper/PX5KFZ3Q"},"agent_actions":{"view_html":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL","download_json":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL.json","view_paper":"https://pith.science/paper/PX5KFZ3Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.04570&json=true","fetch_graph":"https://pith.science/api/pith-number/PX5KFZ3QOW7ATUGBXTOQYBODOL/graph.json","fetch_events":"https://pith.science/api/pith-number/PX5KFZ3QOW7ATUGBXTOQYBODOL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL/action/storage_attestation","attest_author":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL/action/author_attestation","sign_citation":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL/action/citation_signature","submit_replication":"https://pith.science/pith/PX5KFZ3QOW7ATUGBXTOQYBODOL/action/replication_record"}},"created_at":"2026-07-05T04:10:50.612770+00:00","updated_at":"2026-07-05T04:10:50.612770+00:00"}