{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TTWUHLVKZXCTRWVWNHG5CE6NPU","short_pith_number":"pith:TTWUHLVK","schema_version":"1.0","canonical_sha256":"9ced43aeaacdc538dab669cdd113cd7d039922a7937a63f82caaec4bb7b7a862","source":{"kind":"arxiv","id":"2406.17328","version":3},"attestation_state":"computed","paper":{"title":"Dual-Space Knowledge Distillation for Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jinan Xu, Songming Zhang, Xue Zhang, Yufeng Chen, Zengkui Sun","submitted_at":"2024-06-25T07:25:15Z","abstract_excerpt":"Knowledge distillation (KD) is known as a promising solution to compress large language models (LLMs) via transferring their knowledge to smaller models. During this process, white-box KD methods usually minimize the distance between the output distributions of the two models so that more knowledge can be transferred. However, in the current white-box KD framework, the output distributions are from the respective output spaces of the two models, using their own prediction heads. We argue that the space discrepancy will lead to low similarity between the teacher model and the student model on b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17328","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-25T07:25:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d0956cba41168de3ea394b3c44cdf7417508263f2b2732d614e5f4efb6b3fe9d","abstract_canon_sha256":"887912831ea57d480f2876b65847a01ae5702d2eda0c986a3866f39663fa98fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:21.279178Z","signature_b64":"DLKulZmV3cLwuWKIl+kol9RtdhhZkO1jl/7S2A+TmjiIuHN0JJ7ec7rfh27eIKSK7dbJkQK8Ff20PO+K6nRtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ced43aeaacdc538dab669cdd113cd7d039922a7937a63f82caaec4bb7b7a862","last_reissued_at":"2026-07-05T09:14:21.278787Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:21.278787Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dual-Space Knowledge Distillation for Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jinan Xu, Songming Zhang, Xue Zhang, Yufeng Chen, Zengkui Sun","submitted_at":"2024-06-25T07:25:15Z","abstract_excerpt":"Knowledge distillation (KD) is known as a promising solution to compress large language models (LLMs) via transferring their knowledge to smaller models. During this process, white-box KD methods usually minimize the distance between the output distributions of the two models so that more knowledge can be transferred. However, in the current white-box KD framework, the output distributions are from the respective output spaces of the two models, using their own prediction heads. We argue that the space discrepancy will lead to low similarity between the teacher model and the student model on b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17328","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17328/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17328","created_at":"2026-07-05T09:14:21.278841+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17328v3","created_at":"2026-07-05T09:14:21.278841+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17328","created_at":"2026-07-05T09:14:21.278841+00:00"},{"alias_kind":"pith_short_12","alias_value":"TTWUHLVKZXCT","created_at":"2026-07-05T09:14:21.278841+00:00"},{"alias_kind":"pith_short_16","alias_value":"TTWUHLVKZXCTRWVW","created_at":"2026-07-05T09:14:21.278841+00:00"},{"alias_kind":"pith_short_8","alias_value":"TTWUHLVK","created_at":"2026-07-05T09:14:21.278841+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.14629","citing_title":"Switch-KD: Visual-Switch Knowledge Distillation for Vision-Language Models","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU","json":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU.json","graph_json":"https://pith.science/api/pith-number/TTWUHLVKZXCTRWVWNHG5CE6NPU/graph.json","events_json":"https://pith.science/api/pith-number/TTWUHLVKZXCTRWVWNHG5CE6NPU/events.json","paper":"https://pith.science/paper/TTWUHLVK"},"agent_actions":{"view_html":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU","download_json":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU.json","view_paper":"https://pith.science/paper/TTWUHLVK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17328&json=true","fetch_graph":"https://pith.science/api/pith-number/TTWUHLVKZXCTRWVWNHG5CE6NPU/graph.json","fetch_events":"https://pith.science/api/pith-number/TTWUHLVKZXCTRWVWNHG5CE6NPU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU/action/storage_attestation","attest_author":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU/action/author_attestation","sign_citation":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU/action/citation_signature","submit_replication":"https://pith.science/pith/TTWUHLVKZXCTRWVWNHG5CE6NPU/action/replication_record"}},"created_at":"2026-07-05T09:14:21.278841+00:00","updated_at":"2026-07-05T09:14:21.278841+00:00"}