{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XJBBNVVWSTMB7YGKQTDODPM6RE","short_pith_number":"pith:XJBBNVVW","schema_version":"1.0","canonical_sha256":"ba4216d6b694d81fe0ca84c6e1bd9e8918183a7f396241be09177fb220e78ffa","source":{"kind":"arxiv","id":"2012.09816","version":3},"attestation_state":"computed","paper":{"title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Yuanzhi Li, Zeyuan Allen-Zhu","submitted_at":"2020-12-17T18:34:45Z","abstract_excerpt":"We formally study how ensemble of deep learning models can improve test accuracy, and how the superior performance of ensemble can be distilled into a single model using knowledge distillation. We consider the challenging case where the ensemble is simply an average of the outputs of a few independently trained neural networks with the SAME architecture, trained using the SAME algorithm on the SAME data set, and they only differ by the random seeds used in the initialization.\n  We show that ensemble/knowledge distillation in Deep Learning works very differently from traditional learning theory"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.09816","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-17T18:34:45Z","cross_cats_sorted":["cs.NE","math.OC","stat.ML"],"title_canon_sha256":"8f013ec81032165aff59e014868038faac29556c38f381019e701efe4ff61980","abstract_canon_sha256":"4fc92c7c77c321d2e375bd6f083267cc522e96ccbb4e05f3b2bd2596f4b965c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:41:59.930546Z","signature_b64":"xM1UlP0b4PPpNxX7G+Nafg+reZd2grPs2pmR8Z+K2qgyZYv8mZeW4R5i3vzeJBdFPmR66Wm35bSs0451ESosBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba4216d6b694d81fe0ca84c6e1bd9e8918183a7f396241be09177fb220e78ffa","last_reissued_at":"2026-07-05T05:41:59.930124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:41:59.930124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Yuanzhi Li, Zeyuan Allen-Zhu","submitted_at":"2020-12-17T18:34:45Z","abstract_excerpt":"We formally study how ensemble of deep learning models can improve test accuracy, and how the superior performance of ensemble can be distilled into a single model using knowledge distillation. We consider the challenging case where the ensemble is simply an average of the outputs of a few independently trained neural networks with the SAME architecture, trained using the SAME algorithm on the SAME data set, and they only differ by the random seeds used in the initialization.\n  We show that ensemble/knowledge distillation in Deep Learning works very differently from traditional learning theory"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.09816","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.09816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.09816","created_at":"2026-07-05T05:41:59.930180+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.09816v3","created_at":"2026-07-05T05:41:59.930180+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.09816","created_at":"2026-07-05T05:41:59.930180+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJBBNVVWSTMB","created_at":"2026-07-05T05:41:59.930180+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJBBNVVWSTMB7YGK","created_at":"2026-07-05T05:41:59.930180+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJBBNVVW","created_at":"2026-07-05T05:41:59.930180+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23364","citing_title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09658","citing_title":"Muon Learns More Robust and Transferable Features than Adam","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08252","citing_title":"Quantifying and Defending against the Privacy Risk in Logit-based Federated Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2305.07759","citing_title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2203.04153","citing_title":"Easy Ensemble: Simple Deep Ensemble Learning for Sensor-Based Human Activity Recognition","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00901","citing_title":"Provable Knowledge Acquisition and Extraction in One-Layer Transformers","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2401.01335","citing_title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04038","citing_title":"FLAME: Condensing Ensemble Diversity into a Single Network for Efficient Sequential Recommendation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11414","citing_title":"Generative Diffusion Prior Distillation for Long-Context Knowledge Transfer","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08292","citing_title":"Hierarchical Mixture-of-Experts with Two-Stage Optimization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19724","citing_title":"Benign Overfitting in Adversarial Training for Vision Transformers","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07244","citing_title":"Experience Sharing in Mutual Reinforcement Learning for Heterogeneous Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE","json":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE.json","graph_json":"https://pith.science/api/pith-number/XJBBNVVWSTMB7YGKQTDODPM6RE/graph.json","events_json":"https://pith.science/api/pith-number/XJBBNVVWSTMB7YGKQTDODPM6RE/events.json","paper":"https://pith.science/paper/XJBBNVVW"},"agent_actions":{"view_html":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE","download_json":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE.json","view_paper":"https://pith.science/paper/XJBBNVVW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.09816&json=true","fetch_graph":"https://pith.science/api/pith-number/XJBBNVVWSTMB7YGKQTDODPM6RE/graph.json","fetch_events":"https://pith.science/api/pith-number/XJBBNVVWSTMB7YGKQTDODPM6RE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE/action/storage_attestation","attest_author":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE/action/author_attestation","sign_citation":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE/action/citation_signature","submit_replication":"https://pith.science/pith/XJBBNVVWSTMB7YGKQTDODPM6RE/action/replication_record"}},"created_at":"2026-07-05T05:41:59.930180+00:00","updated_at":"2026-07-05T05:41:59.930180+00:00"}