{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3Z6ZFPYJIGTZL23JGJIZYXC6DR","short_pith_number":"pith:3Z6ZFPYJ","schema_version":"1.0","canonical_sha256":"de7d92bf0941a795eb6932519c5c5e1c5d1e0d359bdb17e8871a4d3f559c3fb6","source":{"kind":"arxiv","id":"2306.13649","version":3},"attestation_state":"computed","paper":{"title":"On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Matthieu Geist, Nino Vieillard, Olivier Bachem, Piotr Stanczyk, Rishabh Agarwal, Sabela Ramos, Yongchao Zhou","submitted_at":"2023-06-23T17:56:26Z","abstract_excerpt":"Knowledge distillation (KD) is widely used for compressing a teacher model to reduce its inference cost and memory footprint, by training a smaller student model. However, current KD methods for auto-regressive sequence models suffer from distribution mismatch between output sequences seen during training and those generated by the student during inference. To address this issue, we introduce Generalized Knowledge Distillation (GKD). Instead of solely relying on a fixed set of output sequences, GKD trains the student on its self-generated output sequences by leveraging feedback from the teache"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.13649","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-23T17:56:26Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a1e839e0ebb01d6ed89cbf56a10842de6baaa7a05209b87ef5f411622e0fe35d","abstract_canon_sha256":"389ee6658f57e9b1a62394a4d26a57ea927da59024257f05bf61fe567e9be085"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:34:15.598853Z","signature_b64":"nuzDoT0ZMNGfXIUFoyddMOx6SIyC+sqTnu0dMdSPmwalbmxW4ALJnQuYzS0za1nb4xp7x6yhigv8lbT8UTRLDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de7d92bf0941a795eb6932519c5c5e1c5d1e0d359bdb17e8871a4d3f559c3fb6","last_reissued_at":"2026-07-05T07:34:15.598455Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:34:15.598455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Matthieu Geist, Nino Vieillard, Olivier Bachem, Piotr Stanczyk, Rishabh Agarwal, Sabela Ramos, Yongchao Zhou","submitted_at":"2023-06-23T17:56:26Z","abstract_excerpt":"Knowledge distillation (KD) is widely used for compressing a teacher model to reduce its inference cost and memory footprint, by training a smaller student model. However, current KD methods for auto-regressive sequence models suffer from distribution mismatch between output sequences seen during training and those generated by the student during inference. To address this issue, we introduce Generalized Knowledge Distillation (GKD). Instead of solely relying on a fixed set of output sequences, GKD trains the student on its self-generated output sequences by leveraging feedback from the teache"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13649","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.13649/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.13649","created_at":"2026-07-05T07:34:15.598510+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.13649v3","created_at":"2026-07-05T07:34:15.598510+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13649","created_at":"2026-07-05T07:34:15.598510+00:00"},{"alias_kind":"pith_short_12","alias_value":"3Z6ZFPYJIGTZ","created_at":"2026-07-05T07:34:15.598510+00:00"},{"alias_kind":"pith_short_16","alias_value":"3Z6ZFPYJIGTZL23J","created_at":"2026-07-05T07:34:15.598510+00:00"},{"alias_kind":"pith_short_8","alias_value":"3Z6ZFPYJ","created_at":"2026-07-05T07:34:15.598510+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":36,"internal_anchor_count":4,"sample":[{"citing_arxiv_id":"2607.07050","citing_title":"When Top-K Misses the Decision: Tool-Call Drift in Multi-Teacher On-Policy Distillation","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05804","citing_title":"TurnOPD: Making On-Policy Distillation Turn-Aware for Efficient Long-Horizon Agent Training","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05339","citing_title":"TREK: Distill to Explore, Reinforce to Refine","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05394","citing_title":"Weak-to-Strong Generalization via Direct On-Policy Distillation","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24747","citing_title":"Scaling Laws for Task-Specific LLM Distillation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02234","citing_title":"Purified OPSD: On-Policy Self-Distillation Without Losing How to Think","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12507","citing_title":"Rubric-Guided Self-Distillation: Post-Training Without Rubric Verifiers","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10334","citing_title":"Self-Distillation Policy Optimization via Visual Feedback: Bridging Code and Visual Artifacts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07082","citing_title":"On the Geometry of On-Policy Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03532","citing_title":"When Should the Teacher Move? Temporal Coupling and Stability in Self On-Policy Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03089","citing_title":"Constitutional On-Policy Safe Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12634","citing_title":"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30383","citing_title":"Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29340","citing_title":"PHF: Privileged Hidden Flow for On-Policy Self-Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25378","citing_title":"CollectionLoRA: Collecting 50 Effects in 1 LoRA via Multi-Teacher On-Policy Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27095","citing_title":"Adversarial Dual On-Policy Distillation from Expressive Teacher","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29247","citing_title":"DenseSteer: Steering Small Language Models towards Dense Math Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00147","citing_title":"RAFT: Data Refinement and Adaptive Distillation for Domain Fine-Tuning with Alleviated Forgetting","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14084","citing_title":"TIP: Token Importance in On-Policy Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11739","citing_title":"Learning to Foresee: Unveiling the Unlocking Efficiency of On-Policy Distillation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18141","citing_title":"A Brief Overview: On-Policy Self-Distillation In Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21606","citing_title":"When Are Teacher Tokens Reliable? Position-Weighted On-Policy Self-Distillation for Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14084","citing_title":"TIP: Token Importance in On-Policy Distillation","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR","json":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR.json","graph_json":"https://pith.science/api/pith-number/3Z6ZFPYJIGTZL23JGJIZYXC6DR/graph.json","events_json":"https://pith.science/api/pith-number/3Z6ZFPYJIGTZL23JGJIZYXC6DR/events.json","paper":"https://pith.science/paper/3Z6ZFPYJ"},"agent_actions":{"view_html":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR","download_json":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR.json","view_paper":"https://pith.science/paper/3Z6ZFPYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.13649&json=true","fetch_graph":"https://pith.science/api/pith-number/3Z6ZFPYJIGTZL23JGJIZYXC6DR/graph.json","fetch_events":"https://pith.science/api/pith-number/3Z6ZFPYJIGTZL23JGJIZYXC6DR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR/action/storage_attestation","attest_author":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR/action/author_attestation","sign_citation":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR/action/citation_signature","submit_replication":"https://pith.science/pith/3Z6ZFPYJIGTZL23JGJIZYXC6DR/action/replication_record"}},"created_at":"2026-07-05T07:34:15.598510+00:00","updated_at":"2026-07-05T07:34:15.598510+00:00"}