{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3KGUE2DY2E5IDE7OVV7CDSGHDA","short_pith_number":"pith:3KGUE2DY","schema_version":"1.0","canonical_sha256":"da8d426878d13a8193eead7e21c8c71824d6153cb7f4feceb03e804d4aa18ae1","source":{"kind":"arxiv","id":"2305.15032","version":1},"attestation_state":"computed","paper":{"title":"How to Distill your BERT: An Empirical Study on the Impact of Weight Initialisation and Distillation Objectives","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Hinrich Sch\\\"utze, Leonie Weissweiler, Xinpeng Wang","submitted_at":"2023-05-24T11:16:09Z","abstract_excerpt":"Recently, various intermediate layer distillation (ILD) objectives have been shown to improve compression of BERT models via Knowledge Distillation (KD). However, a comprehensive evaluation of the objectives in both task-specific and task-agnostic settings is lacking. To the best of our knowledge, this is the first work comprehensively evaluating distillation objectives in both settings. We show that attention transfer gives the best performance overall. We also study the impact of layer choice when initializing the student from the teacher layers, finding a significant impact on the performan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.15032","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T11:16:09Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"0d8890275d577717c6d916a7940a58c171d44200650bb0578f91135355903890","abstract_canon_sha256":"51974ca4f56e137234fda2177bb84ddf6716a51a93b471627d72074376f489ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:30.252025Z","signature_b64":"t+87PA0bOk2K2yb3awfmjniXKoAAifjBZc44/XjjXmiFMYqoJw/U43sD7V6aNkmSbqs22MSLB/qlJIr9sA1cCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da8d426878d13a8193eead7e21c8c71824d6153cb7f4feceb03e804d4aa18ae1","last_reissued_at":"2026-07-05T06:13:30.251610Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:30.251610Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Distill your BERT: An Empirical Study on the Impact of Weight Initialisation and Distillation Objectives","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Hinrich Sch\\\"utze, Leonie Weissweiler, Xinpeng Wang","submitted_at":"2023-05-24T11:16:09Z","abstract_excerpt":"Recently, various intermediate layer distillation (ILD) objectives have been shown to improve compression of BERT models via Knowledge Distillation (KD). However, a comprehensive evaluation of the objectives in both task-specific and task-agnostic settings is lacking. To the best of our knowledge, this is the first work comprehensively evaluating distillation objectives in both settings. We show that attention transfer gives the best performance overall. We also study the impact of layer choice when initializing the student from the teacher layers, finding a significant impact on the performan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.15032","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.15032/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.15032","created_at":"2026-07-05T06:13:30.251665+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.15032v1","created_at":"2026-07-05T06:13:30.251665+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.15032","created_at":"2026-07-05T06:13:30.251665+00:00"},{"alias_kind":"pith_short_12","alias_value":"3KGUE2DY2E5I","created_at":"2026-07-05T06:13:30.251665+00:00"},{"alias_kind":"pith_short_16","alias_value":"3KGUE2DY2E5IDE7O","created_at":"2026-07-05T06:13:30.251665+00:00"},{"alias_kind":"pith_short_8","alias_value":"3KGUE2DY","created_at":"2026-07-05T06:13:30.251665+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.13825","citing_title":"Feature Alignment and Representation Transfer in Knowledge Distillation for Large Language Models","ref_index":82,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA","json":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA.json","graph_json":"https://pith.science/api/pith-number/3KGUE2DY2E5IDE7OVV7CDSGHDA/graph.json","events_json":"https://pith.science/api/pith-number/3KGUE2DY2E5IDE7OVV7CDSGHDA/events.json","paper":"https://pith.science/paper/3KGUE2DY"},"agent_actions":{"view_html":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA","download_json":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA.json","view_paper":"https://pith.science/paper/3KGUE2DY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.15032&json=true","fetch_graph":"https://pith.science/api/pith-number/3KGUE2DY2E5IDE7OVV7CDSGHDA/graph.json","fetch_events":"https://pith.science/api/pith-number/3KGUE2DY2E5IDE7OVV7CDSGHDA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA/action/storage_attestation","attest_author":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA/action/author_attestation","sign_citation":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA/action/citation_signature","submit_replication":"https://pith.science/pith/3KGUE2DY2E5IDE7OVV7CDSGHDA/action/replication_record"}},"created_at":"2026-07-05T06:13:30.251665+00:00","updated_at":"2026-07-05T06:13:30.251665+00:00"}