{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:ES67U3CF5G6U4I4DBPPQDNYLTR","short_pith_number":"pith:ES67U3CF","schema_version":"1.0","canonical_sha256":"24bdfa6c45e9bd4e23830bdf01b70b9c66418ad83f46e8aefdede0a9d6db1761","source":{"kind":"arxiv","id":"2002.03532","version":2},"attestation_state":"computed","paper":{"title":"Understanding and Improving Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anima Singh, Dong Lin, Ed H. Chi, Jiaxi Tang, Rakesh Shivanna, Sagar Jain, Zhe Zhao","submitted_at":"2020-02-10T04:21:41Z","abstract_excerpt":"Knowledge Distillation (KD) is a model-agnostic technique to improve model quality while having a fixed capacity budget. It is a commonly used technique for model compression, where a larger capacity teacher model with better quality is used to train a more compact student model with better inference efficiency. Through distillation, one hopes to benefit from student's compactness, without sacrificing too much on model quality. Despite the large success of knowledge distillation, better understanding of how it benefits student model's training dynamics remains under-explored. In this paper, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.03532","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-10T04:21:41Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"2e4d3c09f2ede7fb3d44c7fb39aba30ef7182473df5acea94f87d573e04e66b6","abstract_canon_sha256":"f752b779cce28e8899096e4ad64c3afd1ac6de88de2c0585acea85117ab1a39a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:18:50.740810Z","signature_b64":"NijDOc0XCFQAxdUpJFpbLXmv4oa/OGX3Mo51zWuhmDB9ZrwDGS7h1P/t/EKNQBIaCude+AvQC4PXzGSwL7DACQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24bdfa6c45e9bd4e23830bdf01b70b9c66418ad83f46e8aefdede0a9d6db1761","last_reissued_at":"2026-07-05T02:18:50.740445Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:18:50.740445Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding and Improving Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anima Singh, Dong Lin, Ed H. Chi, Jiaxi Tang, Rakesh Shivanna, Sagar Jain, Zhe Zhao","submitted_at":"2020-02-10T04:21:41Z","abstract_excerpt":"Knowledge Distillation (KD) is a model-agnostic technique to improve model quality while having a fixed capacity budget. It is a commonly used technique for model compression, where a larger capacity teacher model with better quality is used to train a more compact student model with better inference efficiency. Through distillation, one hopes to benefit from student's compactness, without sacrificing too much on model quality. Despite the large success of knowledge distillation, better understanding of how it benefits student model's training dynamics remains under-explored. In this paper, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.03532","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.03532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.03532","created_at":"2026-07-05T02:18:50.740504+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.03532v2","created_at":"2026-07-05T02:18:50.740504+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.03532","created_at":"2026-07-05T02:18:50.740504+00:00"},{"alias_kind":"pith_short_12","alias_value":"ES67U3CF5G6U","created_at":"2026-07-05T02:18:50.740504+00:00"},{"alias_kind":"pith_short_16","alias_value":"ES67U3CF5G6U4I4D","created_at":"2026-07-05T02:18:50.740504+00:00"},{"alias_kind":"pith_short_8","alias_value":"ES67U3CF","created_at":"2026-07-05T02:18:50.740504+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03052","citing_title":"What Do Students Learn? A Feature-Level Analysis of Dark Knowledge","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00306","citing_title":"Rethinking the Role of Temperature in Large Language Model Distillation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23857","citing_title":"Strong Teacher Not Needed? On Distillation in LLM Pretraining","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2112.11447","citing_title":"Multi-Modality Distillation via Learning the teacher's modality-level Gram Matrix","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10861","citing_title":"Knowledge Distillation in Federated Learning: a Survey on Long Lasting Challenges and New Solutions","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11419","citing_title":"Knowledge Distillation for Sensing-Assisted Long-Term Beam Tracking in mmWave Communications","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20357","citing_title":"Consistently Informative Soft-Label Temperature for Knowledge Distillation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18702","citing_title":"Distilling Tabular Foundation Models for Structured Health Data","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11178","citing_title":"PACED: Distillation and On-Policy Self-Distillation at the Frontier of Student Competence","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25786","citing_title":"Homogeneous Stellar Parameters from Heterogeneous Spectra with Deep Learning","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04569","citing_title":"LIVEditor-14B: Lightning Unified Video Editing via In-Context Sparse Attention","ref_index":148,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR","json":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR.json","graph_json":"https://pith.science/api/pith-number/ES67U3CF5G6U4I4DBPPQDNYLTR/graph.json","events_json":"https://pith.science/api/pith-number/ES67U3CF5G6U4I4DBPPQDNYLTR/events.json","paper":"https://pith.science/paper/ES67U3CF"},"agent_actions":{"view_html":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR","download_json":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR.json","view_paper":"https://pith.science/paper/ES67U3CF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.03532&json=true","fetch_graph":"https://pith.science/api/pith-number/ES67U3CF5G6U4I4DBPPQDNYLTR/graph.json","fetch_events":"https://pith.science/api/pith-number/ES67U3CF5G6U4I4DBPPQDNYLTR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR/action/storage_attestation","attest_author":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR/action/author_attestation","sign_citation":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR/action/citation_signature","submit_replication":"https://pith.science/pith/ES67U3CF5G6U4I4DBPPQDNYLTR/action/replication_record"}},"created_at":"2026-07-05T02:18:50.740504+00:00","updated_at":"2026-07-05T02:18:50.740504+00:00"}