{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:UJOMQZJKSVX4EQYUIS7ABHMFON","short_pith_number":"pith:UJOMQZJK","schema_version":"1.0","canonical_sha256":"a25cc8652a956fc2431444be009d85737a3c7e1021c6cf406d6a662a839e0234","source":{"kind":"arxiv","id":"2106.05237","version":2},"attestation_state":"computed","paper":{"title":"Knowledge distillation: A good teacher is patient and consistent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Am\\'elie Royer, Larisa Markeeva, Lucas Beyer, Rohan Anil, Xiaohua Zhai","submitted_at":"2021-06-09T17:20:40Z","abstract_excerpt":"There is a growing discrepancy in computer vision between large-scale models that achieve state-of-the-art performance and models that are affordable in practical applications. In this paper we address this issue and significantly bridge the gap between these two types of models. Throughout our empirical investigation we do not aim to necessarily propose a new method, but strive to identify a robust and effective recipe for making state-of-the-art large scale models affordable in practice. We demonstrate that, when performed correctly, knowledge distillation can be a powerful tool for reducing"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.05237","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-09T17:20:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"674e910326b902b816f30c569b5c8bcd371175de0b4da1a72c3c29618a9c3c93","abstract_canon_sha256":"553d85781e50acfbf4628a10a6cba32819d3dedeaa2adf53b5f466726a0288bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:33:19.739195Z","signature_b64":"VEBqUrbf1YwY7STX92lC20XEWI7l6cJW2q8NJTN4LZSu5UpQjgyeMbw8uUVBEM7bbnQSzyvZAFHOB3u+gZweBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a25cc8652a956fc2431444be009d85737a3c7e1021c6cf406d6a662a839e0234","last_reissued_at":"2026-07-05T04:33:19.738714Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:33:19.738714Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Knowledge distillation: A good teacher is patient and consistent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Am\\'elie Royer, Larisa Markeeva, Lucas Beyer, Rohan Anil, Xiaohua Zhai","submitted_at":"2021-06-09T17:20:40Z","abstract_excerpt":"There is a growing discrepancy in computer vision between large-scale models that achieve state-of-the-art performance and models that are affordable in practical applications. In this paper we address this issue and significantly bridge the gap between these two types of models. Throughout our empirical investigation we do not aim to necessarily propose a new method, but strive to identify a robust and effective recipe for making state-of-the-art large scale models affordable in practice. We demonstrate that, when performed correctly, knowledge distillation can be a powerful tool for reducing"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.05237","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.05237/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.05237","created_at":"2026-07-05T04:33:19.738777+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.05237v2","created_at":"2026-07-05T04:33:19.738777+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.05237","created_at":"2026-07-05T04:33:19.738777+00:00"},{"alias_kind":"pith_short_12","alias_value":"UJOMQZJKSVX4","created_at":"2026-07-05T04:33:19.738777+00:00"},{"alias_kind":"pith_short_16","alias_value":"UJOMQZJKSVX4EQYU","created_at":"2026-07-05T04:33:19.738777+00:00"},{"alias_kind":"pith_short_8","alias_value":"UJOMQZJK","created_at":"2026-07-05T04:33:19.738777+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24747","citing_title":"Scaling Laws for Task-Specific LLM Distillation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24007","citing_title":"Fast and Slow Variational Continual Learning","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15689","citing_title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON","json":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON.json","graph_json":"https://pith.science/api/pith-number/UJOMQZJKSVX4EQYUIS7ABHMFON/graph.json","events_json":"https://pith.science/api/pith-number/UJOMQZJKSVX4EQYUIS7ABHMFON/events.json","paper":"https://pith.science/paper/UJOMQZJK"},"agent_actions":{"view_html":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON","download_json":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON.json","view_paper":"https://pith.science/paper/UJOMQZJK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.05237&json=true","fetch_graph":"https://pith.science/api/pith-number/UJOMQZJKSVX4EQYUIS7ABHMFON/graph.json","fetch_events":"https://pith.science/api/pith-number/UJOMQZJKSVX4EQYUIS7ABHMFON/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON/action/storage_attestation","attest_author":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON/action/author_attestation","sign_citation":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON/action/citation_signature","submit_replication":"https://pith.science/pith/UJOMQZJKSVX4EQYUIS7ABHMFON/action/replication_record"}},"created_at":"2026-07-05T04:33:19.738777+00:00","updated_at":"2026-07-05T04:33:19.738777+00:00"}