{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y3ML34OSVLSW5DQ4QJ7E53CBO4","short_pith_number":"pith:Y3ML34OS","schema_version":"1.0","canonical_sha256":"c6d8bdf1d2aae56e8e1c827e4eec4177186707b7a96f6d1dcf081f6b390f043b","source":{"kind":"arxiv","id":"2406.19774","version":2},"attestation_state":"computed","paper":{"title":"Direct Preference Knowledge Distillation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dequan Wang, Furu Wei, Li Dong, Yixing Li, Yu Cheng, Yuxian Gu","submitted_at":"2024-06-28T09:23:40Z","abstract_excerpt":"In the field of large language models (LLMs), Knowledge Distillation (KD) is a critical technique for transferring capabilities from teacher models to student models. However, existing KD methods face limitations and challenges in distillation of LLMs, including efficiency and insufficient measurement capabilities of traditional KL divergence. It is shown that LLMs can serve as an implicit reward function, which we define as a supplement to KL divergence. In this work, we propose Direct Preference Knowledge Distillation (DPKD) for LLMs. DPKD utilizes distribution divergence to represent the pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19774","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-28T09:23:40Z","cross_cats_sorted":[],"title_canon_sha256":"67a0320d34d5d9b0af771c81792ba8005e028e8b1943837fba8a68d2f56accb3","abstract_canon_sha256":"c1ad1b0ca854fdd66f5bebc4502c6b184fc1adb670c1ffb9d255f8d4cd10b054"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:09.621165Z","signature_b64":"4dHu2Cq8kPAhUPFJoPXhgvH0PWLriaHwnFhuUmX7egSfe4EcovIfUBmU7Z9iLC+59FV4ABlQvLxnPV/qYBhhAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c6d8bdf1d2aae56e8e1c827e4eec4177186707b7a96f6d1dcf081f6b390f043b","last_reissued_at":"2026-07-05T10:45:09.620626Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:09.620626Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct Preference Knowledge Distillation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dequan Wang, Furu Wei, Li Dong, Yixing Li, Yu Cheng, Yuxian Gu","submitted_at":"2024-06-28T09:23:40Z","abstract_excerpt":"In the field of large language models (LLMs), Knowledge Distillation (KD) is a critical technique for transferring capabilities from teacher models to student models. However, existing KD methods face limitations and challenges in distillation of LLMs, including efficiency and insufficient measurement capabilities of traditional KL divergence. It is shown that LLMs can serve as an implicit reward function, which we define as a supplement to KL divergence. In this work, we propose Direct Preference Knowledge Distillation (DPKD) for LLMs. DPKD utilizes distribution divergence to represent the pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19774","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19774/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19774","created_at":"2026-07-05T10:45:09.620686+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19774v2","created_at":"2026-07-05T10:45:09.620686+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19774","created_at":"2026-07-05T10:45:09.620686+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y3ML34OSVLSW","created_at":"2026-07-05T10:45:09.620686+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y3ML34OSVLSW5DQ4","created_at":"2026-07-05T10:45:09.620686+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y3ML34OS","created_at":"2026-07-05T10:45:09.620686+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29869","citing_title":"ARKD: Adaptive Reinforcement Learning-Guided Bidirectional KL Divergence Distillation for Text Generation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22942","citing_title":"Understanding Knowledge Distillation in Post-Training: When It Helps and When It Fails","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09741","citing_title":"ExecTune: Effective Steering of Black-Box LLMs with Guide Models","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4","json":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4.json","graph_json":"https://pith.science/api/pith-number/Y3ML34OSVLSW5DQ4QJ7E53CBO4/graph.json","events_json":"https://pith.science/api/pith-number/Y3ML34OSVLSW5DQ4QJ7E53CBO4/events.json","paper":"https://pith.science/paper/Y3ML34OS"},"agent_actions":{"view_html":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4","download_json":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4.json","view_paper":"https://pith.science/paper/Y3ML34OS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19774&json=true","fetch_graph":"https://pith.science/api/pith-number/Y3ML34OSVLSW5DQ4QJ7E53CBO4/graph.json","fetch_events":"https://pith.science/api/pith-number/Y3ML34OSVLSW5DQ4QJ7E53CBO4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4/action/storage_attestation","attest_author":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4/action/author_attestation","sign_citation":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4/action/citation_signature","submit_replication":"https://pith.science/pith/Y3ML34OSVLSW5DQ4QJ7E53CBO4/action/replication_record"}},"created_at":"2026-07-05T10:45:09.620686+00:00","updated_at":"2026-07-05T10:45:09.620686+00:00"}