{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:GGRYWHWAIU7HCBR6Q7WFWC2HEJ","short_pith_number":"pith:GGRYWHWA","schema_version":"1.0","canonical_sha256":"31a38b1ec0453e71063e87ec5b0b47227322401a38e8901d5ddec0c8f34bbbc5","source":{"kind":"arxiv","id":"2206.06661","version":2},"attestation_state":"computed","paper":{"title":"Toward Student-Oriented Teacher Network Training For Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chengyu Dong, Jingbo Shang, Liyuan Liu","submitted_at":"2022-06-14T07:51:25Z","abstract_excerpt":"How to conduct teacher training for knowledge distillation is still an open problem. It has been widely observed that a best-performing teacher does not necessarily yield the best-performing student, suggesting a fundamental discrepancy between the current teacher training practice and the ideal teacher training strategy. To fill this gap, we explore the feasibility of training a teacher that is oriented toward student performance with empirical risk minimization (ERM). Our analyses are inspired by the recent findings that the effectiveness of knowledge distillation hinges on the teacher's cap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.06661","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-06-14T07:51:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9939fcfff6c5970d0b341b7cd05723694f9567e7bbb646070b0048037798ff8c","abstract_canon_sha256":"7964c908a210037d76bc385a5369ce19865877a6d3e34ab21e7eac9f9d1b5cb3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:17:08.235594Z","signature_b64":"oqvNVDjkeYsY4JksimyxJTfLjQRskIbvCtDnv8p1P/paKXgYFM5G/xv8CqhMcynIICfzhGSrWCvkOYI/14jZDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31a38b1ec0453e71063e87ec5b0b47227322401a38e8901d5ddec0c8f34bbbc5","last_reissued_at":"2026-07-05T08:17:08.235112Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:17:08.235112Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Toward Student-Oriented Teacher Network Training For Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chengyu Dong, Jingbo Shang, Liyuan Liu","submitted_at":"2022-06-14T07:51:25Z","abstract_excerpt":"How to conduct teacher training for knowledge distillation is still an open problem. It has been widely observed that a best-performing teacher does not necessarily yield the best-performing student, suggesting a fundamental discrepancy between the current teacher training practice and the ideal teacher training strategy. To fill this gap, we explore the feasibility of training a teacher that is oriented toward student performance with empirical risk minimization (ERM). Our analyses are inspired by the recent findings that the effectiveness of knowledge distillation hinges on the teacher's cap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.06661","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.06661/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.06661","created_at":"2026-07-05T08:17:08.235171+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.06661v2","created_at":"2026-07-05T08:17:08.235171+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.06661","created_at":"2026-07-05T08:17:08.235171+00:00"},{"alias_kind":"pith_short_12","alias_value":"GGRYWHWAIU7H","created_at":"2026-07-05T08:17:08.235171+00:00"},{"alias_kind":"pith_short_16","alias_value":"GGRYWHWAIU7HCBR6","created_at":"2026-07-05T08:17:08.235171+00:00"},{"alias_kind":"pith_short_8","alias_value":"GGRYWHWA","created_at":"2026-07-05T08:17:08.235171+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23041","citing_title":"ReMem: Mutual Information-Aware Fine-tuning of Pretrained Vision Transformers for Effective Knowledge Distillation","ref_index":2023,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ","json":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ.json","graph_json":"https://pith.science/api/pith-number/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/graph.json","events_json":"https://pith.science/api/pith-number/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/events.json","paper":"https://pith.science/paper/GGRYWHWA"},"agent_actions":{"view_html":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ","download_json":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ.json","view_paper":"https://pith.science/paper/GGRYWHWA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.06661&json=true","fetch_graph":"https://pith.science/api/pith-number/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/graph.json","fetch_events":"https://pith.science/api/pith-number/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/action/storage_attestation","attest_author":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/action/author_attestation","sign_citation":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/action/citation_signature","submit_replication":"https://pith.science/pith/GGRYWHWAIU7HCBR6Q7WFWC2HEJ/action/replication_record"}},"created_at":"2026-07-05T08:17:08.235171+00:00","updated_at":"2026-07-05T08:17:08.235171+00:00"}