{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:U4PFJ7RG3ZHH22JGE6S7UA2HS6","short_pith_number":"pith:U4PFJ7RG","schema_version":"1.0","canonical_sha256":"a71e54fe26de4e7d692627a5fa034797a91cdaeec43d37516b734182702c5284","source":{"kind":"arxiv","id":"2110.08532","version":1},"attestation_state":"computed","paper":{"title":"Pro-KD: Progressive Distillation by Following the Footsteps of the Teacher","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ali Ghodsi, Ali Saheb Pasand, Aref Jafari, Mehdi Rezagholizadeh, Pranav Sharma, Puneeth Salad","submitted_at":"2021-10-16T09:49:43Z","abstract_excerpt":"With ever growing scale of neural models, knowledge distillation (KD) attracts more attention as a prominent tool for neural model compression. However, there are counter intuitive observations in the literature showing some challenging limitations of KD. A case in point is that the best performing checkpoint of the teacher might not necessarily be the best teacher for training the student in KD. Therefore, one important question would be how to find the best checkpoint of the teacher for distillation? Searching through the checkpoints of the teacher would be a very tedious and computationally"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.08532","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2021-10-16T09:49:43Z","cross_cats_sorted":[],"title_canon_sha256":"6f517e925acaef3912e9644628798ebe27c3217bb903dd7d28f854b7b3cef9af","abstract_canon_sha256":"df672691e0f973677fc589ca6a87c1a43077f77278b44b380d1225c73042d991"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:23:18.041187Z","signature_b64":"cIpnpSFScmxYaKrwyeXuu/F5OOYXTRKtSaybHhbudKUvn+8xsSVb413R4LfhjFepsvdsQaISge4c/LSIlsFhBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a71e54fe26de4e7d692627a5fa034797a91cdaeec43d37516b734182702c5284","last_reissued_at":"2026-07-05T03:23:18.040785Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:23:18.040785Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pro-KD: Progressive Distillation by Following the Footsteps of the Teacher","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ali Ghodsi, Ali Saheb Pasand, Aref Jafari, Mehdi Rezagholizadeh, Pranav Sharma, Puneeth Salad","submitted_at":"2021-10-16T09:49:43Z","abstract_excerpt":"With ever growing scale of neural models, knowledge distillation (KD) attracts more attention as a prominent tool for neural model compression. However, there are counter intuitive observations in the literature showing some challenging limitations of KD. A case in point is that the best performing checkpoint of the teacher might not necessarily be the best teacher for training the student in KD. Therefore, one important question would be how to find the best checkpoint of the teacher for distillation? Searching through the checkpoints of the teacher would be a very tedious and computationally"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.08532","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.08532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.08532","created_at":"2026-07-05T03:23:18.040840+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.08532v1","created_at":"2026-07-05T03:23:18.040840+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.08532","created_at":"2026-07-05T03:23:18.040840+00:00"},{"alias_kind":"pith_short_12","alias_value":"U4PFJ7RG3ZHH","created_at":"2026-07-05T03:23:18.040840+00:00"},{"alias_kind":"pith_short_16","alias_value":"U4PFJ7RG3ZHH22JG","created_at":"2026-07-05T03:23:18.040840+00:00"},{"alias_kind":"pith_short_8","alias_value":"U4PFJ7RG","created_at":"2026-07-05T03:23:18.040840+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06192","citing_title":"Right Time to Learn:Promoting Generalization via Bio-inspired Spacing Effect in Knowledge Distillation","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6","json":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6.json","graph_json":"https://pith.science/api/pith-number/U4PFJ7RG3ZHH22JGE6S7UA2HS6/graph.json","events_json":"https://pith.science/api/pith-number/U4PFJ7RG3ZHH22JGE6S7UA2HS6/events.json","paper":"https://pith.science/paper/U4PFJ7RG"},"agent_actions":{"view_html":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6","download_json":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6.json","view_paper":"https://pith.science/paper/U4PFJ7RG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.08532&json=true","fetch_graph":"https://pith.science/api/pith-number/U4PFJ7RG3ZHH22JGE6S7UA2HS6/graph.json","fetch_events":"https://pith.science/api/pith-number/U4PFJ7RG3ZHH22JGE6S7UA2HS6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6/action/storage_attestation","attest_author":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6/action/author_attestation","sign_citation":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6/action/citation_signature","submit_replication":"https://pith.science/pith/U4PFJ7RG3ZHH22JGE6S7UA2HS6/action/replication_record"}},"created_at":"2026-07-05T03:23:18.040840+00:00","updated_at":"2026-07-05T03:23:18.040840+00:00"}