{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HCZDY465TMHJSZC6GOZZ63R6TO","short_pith_number":"pith:HCZDY465","schema_version":"1.0","canonical_sha256":"38b23c73dd9b0e99645e33b39f6e3e9b87f3d5375ffdc8de85a8661aa393ed41","source":{"kind":"arxiv","id":"2105.08919","version":1},"attestation_state":"computed","paper":{"title":"Comparing Kullback-Leibler Divergence and Mean Squared Error Loss in Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Jaehoon Oh, Nakyil Kim, Sangwook Cho, Se-Young Yun, Taehyeon Kim","submitted_at":"2021-05-19T04:40:53Z","abstract_excerpt":"Knowledge distillation (KD), transferring knowledge from a cumbersome teacher model to a lightweight student model, has been investigated to design efficient neural architectures. Generally, the objective function of KD is the Kullback-Leibler (KL) divergence loss between the softened probability distributions of the teacher model and the student model with the temperature scaling hyperparameter tau. Despite its widespread use, few studies have discussed the influence of such softening on generalization. Here, we theoretically show that the KL divergence loss focuses on the logit matching when"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.08919","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-19T04:40:53Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"ca2983adad948f4abe421799ad1048bb0b257863b2a1cc0dc1989567368aba32","abstract_canon_sha256":"d4809052645a5b1dcead6ecb456159fbeffec5344e04ecd8bdc8ab318cea8226"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:41:35.145488Z","signature_b64":"65iqWOO+dJEoq6nRy/RjqlcyKN4WmqgHnyrGn1uv9MhU5K1iBnqBpFb6tFsE0mYwedA4T2l2VmSoAqXllS3XDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38b23c73dd9b0e99645e33b39f6e3e9b87f3d5375ffdc8de85a8661aa393ed41","last_reissued_at":"2026-07-05T02:41:35.145052Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:41:35.145052Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comparing Kullback-Leibler Divergence and Mean Squared Error Loss in Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Jaehoon Oh, Nakyil Kim, Sangwook Cho, Se-Young Yun, Taehyeon Kim","submitted_at":"2021-05-19T04:40:53Z","abstract_excerpt":"Knowledge distillation (KD), transferring knowledge from a cumbersome teacher model to a lightweight student model, has been investigated to design efficient neural architectures. Generally, the objective function of KD is the Kullback-Leibler (KL) divergence loss between the softened probability distributions of the teacher model and the student model with the temperature scaling hyperparameter tau. Despite its widespread use, few studies have discussed the influence of such softening on generalization. Here, we theoretically show that the KL divergence loss focuses on the logit matching when"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.08919","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.08919/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.08919","created_at":"2026-07-05T02:41:35.145108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.08919v1","created_at":"2026-07-05T02:41:35.145108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.08919","created_at":"2026-07-05T02:41:35.145108+00:00"},{"alias_kind":"pith_short_12","alias_value":"HCZDY465TMHJ","created_at":"2026-07-05T02:41:35.145108+00:00"},{"alias_kind":"pith_short_16","alias_value":"HCZDY465TMHJSZC6","created_at":"2026-07-05T02:41:35.145108+00:00"},{"alias_kind":"pith_short_8","alias_value":"HCZDY465","created_at":"2026-07-05T02:41:35.145108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03069","citing_title":"ROBUST-WT: Robust Uncertainty-aware Segmentation Transform via Whitening and Training Enhancements","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23645","citing_title":"Learning Through Noise: Why Subliminal Learning Works and When It Fails","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11178","citing_title":"PACED: Distillation and On-Policy Self-Distillation at the Frontier of Student Competence","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.25383","citing_title":"CLIP-RD: Relative Distillation for Efficient CLIP Knowledge Distillation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07207","citing_title":"Direct-to-Event Spiking Neural Network Transfer","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO","json":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO.json","graph_json":"https://pith.science/api/pith-number/HCZDY465TMHJSZC6GOZZ63R6TO/graph.json","events_json":"https://pith.science/api/pith-number/HCZDY465TMHJSZC6GOZZ63R6TO/events.json","paper":"https://pith.science/paper/HCZDY465"},"agent_actions":{"view_html":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO","download_json":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO.json","view_paper":"https://pith.science/paper/HCZDY465","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.08919&json=true","fetch_graph":"https://pith.science/api/pith-number/HCZDY465TMHJSZC6GOZZ63R6TO/graph.json","fetch_events":"https://pith.science/api/pith-number/HCZDY465TMHJSZC6GOZZ63R6TO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO/action/storage_attestation","attest_author":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO/action/author_attestation","sign_citation":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO/action/citation_signature","submit_replication":"https://pith.science/pith/HCZDY465TMHJSZC6GOZZ63R6TO/action/replication_record"}},"created_at":"2026-07-05T02:41:35.145108+00:00","updated_at":"2026-07-05T02:41:35.145108+00:00"}