{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FZXOUAOJTGOBZLAVOJXMC72OOH","short_pith_number":"pith:FZXOUAOJ","schema_version":"1.0","canonical_sha256":"2e6eea01c9999c1cac15726ec17f4e71e5a3bbfbd0cad60259589b770f9207eb","source":{"kind":"arxiv","id":"2301.12923","version":3},"attestation_state":"computed","paper":{"title":"On student-teacher deviations in distillation: does it pay to disobey?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aditya Krishna Menon, Hossein Mobahi, Sanjiv Kumar, Srinadh Bhojanapalli, Vaishnavh Nagarajan","submitted_at":"2023-01-30T14:25:02Z","abstract_excerpt":"Knowledge distillation (KD) has been widely used to improve the test accuracy of a \"student\" network, by training it to mimic the soft probabilities of a trained \"teacher\" network. Yet, it has been shown in recent work that, despite being trained to fit the teacher's probabilities, the student may not only significantly deviate from the teacher probabilities, but may also outdo than the teacher in performance. Our work aims to reconcile this seemingly paradoxical observation. Specifically, we characterize the precise nature of the student-teacher deviations, and argue how they can co-occur wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.12923","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-01-30T14:25:02Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"e9e30ef8c00fe7aec9f96cddd38838d7874e840c4f0856c90391a8a921bffae4","abstract_canon_sha256":"0d1c02eb7504aca0c2ec21d03e26bab0f9d549955bc86b136d81e5d19b1db440"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:57:39.276735Z","signature_b64":"GNC5VCWQsgAB3NUXzENIEE+PlDSJ+Nozm154CjSxM+PDO/UUXX2oKor5UN4uhu42phyYEsh/5fHg4/yJmZNSBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e6eea01c9999c1cac15726ec17f4e71e5a3bbfbd0cad60259589b770f9207eb","last_reissued_at":"2026-07-05T07:57:39.276378Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:57:39.276378Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On student-teacher deviations in distillation: does it pay to disobey?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aditya Krishna Menon, Hossein Mobahi, Sanjiv Kumar, Srinadh Bhojanapalli, Vaishnavh Nagarajan","submitted_at":"2023-01-30T14:25:02Z","abstract_excerpt":"Knowledge distillation (KD) has been widely used to improve the test accuracy of a \"student\" network, by training it to mimic the soft probabilities of a trained \"teacher\" network. Yet, it has been shown in recent work that, despite being trained to fit the teacher's probabilities, the student may not only significantly deviate from the teacher probabilities, but may also outdo than the teacher in performance. Our work aims to reconcile this seemingly paradoxical observation. Specifically, we characterize the precise nature of the student-teacher deviations, and argue how they can co-occur wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.12923","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.12923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.12923","created_at":"2026-07-05T07:57:39.276436+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.12923v3","created_at":"2026-07-05T07:57:39.276436+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.12923","created_at":"2026-07-05T07:57:39.276436+00:00"},{"alias_kind":"pith_short_12","alias_value":"FZXOUAOJTGOB","created_at":"2026-07-05T07:57:39.276436+00:00"},{"alias_kind":"pith_short_16","alias_value":"FZXOUAOJTGOBZLAV","created_at":"2026-07-05T07:57:39.276436+00:00"},{"alias_kind":"pith_short_8","alias_value":"FZXOUAOJ","created_at":"2026-07-05T07:57:39.276436+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.01649","citing_title":"Distilled Pretraining: A modern lens of Data, In-Context Learning and Test-Time Scaling","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH","json":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH.json","graph_json":"https://pith.science/api/pith-number/FZXOUAOJTGOBZLAVOJXMC72OOH/graph.json","events_json":"https://pith.science/api/pith-number/FZXOUAOJTGOBZLAVOJXMC72OOH/events.json","paper":"https://pith.science/paper/FZXOUAOJ"},"agent_actions":{"view_html":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH","download_json":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH.json","view_paper":"https://pith.science/paper/FZXOUAOJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.12923&json=true","fetch_graph":"https://pith.science/api/pith-number/FZXOUAOJTGOBZLAVOJXMC72OOH/graph.json","fetch_events":"https://pith.science/api/pith-number/FZXOUAOJTGOBZLAVOJXMC72OOH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH/action/storage_attestation","attest_author":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH/action/author_attestation","sign_citation":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH/action/citation_signature","submit_replication":"https://pith.science/pith/FZXOUAOJTGOBZLAVOJXMC72OOH/action/replication_record"}},"created_at":"2026-07-05T07:57:39.276436+00:00","updated_at":"2026-07-05T07:57:39.276436+00:00"}