{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WEN4AS73V4X527X66TYB5BMYH7","short_pith_number":"pith:WEN4AS73","schema_version":"1.0","canonical_sha256":"b11bc04bfbaf2fdd7efef4f01e85983fc20631ac8b9d6fc2855b9b1084da3d23","source":{"kind":"arxiv","id":"2409.12512","version":2},"attestation_state":"computed","paper":{"title":"Exploring and Enhancing the Transfer of Distribution in Knowledge Distillation for Autoregressive Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Jing Li, Jun Rao, Liang Ding, Min Zhang, Xuebo Liu, Zepeng Lin","submitted_at":"2024-09-19T07:05:26Z","abstract_excerpt":"Knowledge distillation (KD) is a technique that compresses large teacher models by training smaller student models to mimic them. The success of KD in auto-regressive language models mainly relies on Reverse KL for mode-seeking and student-generated output (SGO) to combat exposure bias. Our theoretical analyses and experimental validation reveal that while Reverse KL effectively mimics certain features of the teacher distribution, it fails to capture most of its behaviors. Conversely, SGO incurs higher computational costs and presents challenges in optimization, particularly when the student m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.12512","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-19T07:05:26Z","cross_cats_sorted":[],"title_canon_sha256":"173d5fd2b05f2d1dfa714b947002d4b25d0bf9f70c20b3e5bfbdb562a3b93d88","abstract_canon_sha256":"a7fa2cc0de1713bc09dc0dac2a23250ee6e555fcce18b979650295af3bd3d8e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:09:26.689852Z","signature_b64":"zUYxOdQv1+fzkNUAZncbQIS4DqnPquMAp5DzXvaWxqTSDNgQ8mLO6Dr58AYbXLPY7TmL+ZuZnyFcRXmPf4hZCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b11bc04bfbaf2fdd7efef4f01e85983fc20631ac8b9d6fc2855b9b1084da3d23","last_reissued_at":"2026-07-05T09:09:26.689441Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:09:26.689441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring and Enhancing the Transfer of Distribution in Knowledge Distillation for Autoregressive Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dacheng Tao, Jing Li, Jun Rao, Liang Ding, Min Zhang, Xuebo Liu, Zepeng Lin","submitted_at":"2024-09-19T07:05:26Z","abstract_excerpt":"Knowledge distillation (KD) is a technique that compresses large teacher models by training smaller student models to mimic them. The success of KD in auto-regressive language models mainly relies on Reverse KL for mode-seeking and student-generated output (SGO) to combat exposure bias. Our theoretical analyses and experimental validation reveal that while Reverse KL effectively mimics certain features of the teacher distribution, it fails to capture most of its behaviors. Conversely, SGO incurs higher computational costs and presents challenges in optimization, particularly when the student m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.12512","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.12512/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.12512","created_at":"2026-07-05T09:09:26.689499+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.12512v2","created_at":"2026-07-05T09:09:26.689499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.12512","created_at":"2026-07-05T09:09:26.689499+00:00"},{"alias_kind":"pith_short_12","alias_value":"WEN4AS73V4X5","created_at":"2026-07-05T09:09:26.689499+00:00"},{"alias_kind":"pith_short_16","alias_value":"WEN4AS73V4X527X6","created_at":"2026-07-05T09:09:26.689499+00:00"},{"alias_kind":"pith_short_8","alias_value":"WEN4AS73","created_at":"2026-07-05T09:09:26.689499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.15303","citing_title":"Self-Evolution Knowledge Distillation for LLM-based Machine Translation","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7","json":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7.json","graph_json":"https://pith.science/api/pith-number/WEN4AS73V4X527X66TYB5BMYH7/graph.json","events_json":"https://pith.science/api/pith-number/WEN4AS73V4X527X66TYB5BMYH7/events.json","paper":"https://pith.science/paper/WEN4AS73"},"agent_actions":{"view_html":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7","download_json":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7.json","view_paper":"https://pith.science/paper/WEN4AS73","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.12512&json=true","fetch_graph":"https://pith.science/api/pith-number/WEN4AS73V4X527X66TYB5BMYH7/graph.json","fetch_events":"https://pith.science/api/pith-number/WEN4AS73V4X527X66TYB5BMYH7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7/action/storage_attestation","attest_author":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7/action/author_attestation","sign_citation":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7/action/citation_signature","submit_replication":"https://pith.science/pith/WEN4AS73V4X527X66TYB5BMYH7/action/replication_record"}},"created_at":"2026-07-05T09:09:26.689499+00:00","updated_at":"2026-07-05T09:09:26.689499+00:00"}