{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2016:TAVOAQQ7MRARLC3LHHIHCGIYKM","short_pith_number":"pith:TAVOAQQ7","schema_version":"1.0","canonical_sha256":"982ae0421f6441158b6b39d0711918531290020eb9769334a374f0a63c07c1ef","source":{"kind":"arxiv","id":"1606.07947","version":4},"attestation_state":"computed","paper":{"title":"Sequence-Level Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Yoon Kim","submitted_at":"2016-06-25T18:16:39Z","abstract_excerpt":"Neural machine translation (NMT) offers a novel alternative formulation of translation that is potentially simpler than statistical approaches. However to reach competitive performance, NMT models need to be exceedingly large. In this paper we consider applying knowledge distillation approaches (Bucila et al., 2006; Hinton et al., 2015) that have proven successful for reducing the size of neural models in other domains to the problem of NMT. We demonstrate that standard knowledge distillation applied to word-level prediction can be effective for NMT, and also introduce two novel sequence-level"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1606.07947","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2016-06-25T18:16:39Z","cross_cats_sorted":["cs.LG","cs.NE"],"title_canon_sha256":"dbabe5f45cb515a088f7cc86ec14544eb690510fcc00e81dedb6c62f539d5aac","abstract_canon_sha256":"349f0b8b2995785f9ca54bd08f648c3b4b8a0e911f70b225db2ea739640d416a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T01:04:05.442663Z","signature_b64":"p1DvZwWHlgx1TqLN1ZIUn1sRWM1n84DzEuUPxisvOtDOgT/3UunqBquZULmyev3ZI47x3xAHG1rcoR4jxChHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"982ae0421f6441158b6b39d0711918531290020eb9769334a374f0a63c07c1ef","last_reissued_at":"2026-05-18T01:04:05.442235Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T01:04:05.442235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sequence-Level Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE"],"primary_cat":"cs.CL","authors_text":"Alexander M. Rush, Yoon Kim","submitted_at":"2016-06-25T18:16:39Z","abstract_excerpt":"Neural machine translation (NMT) offers a novel alternative formulation of translation that is potentially simpler than statistical approaches. However to reach competitive performance, NMT models need to be exceedingly large. In this paper we consider applying knowledge distillation approaches (Bucila et al., 2006; Hinton et al., 2015) that have proven successful for reducing the size of neural models in other domains to the problem of NMT. We demonstrate that standard knowledge distillation applied to word-level prediction can be effective for NMT, and also introduce two novel sequence-level"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1606.07947","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1606.07947","created_at":"2026-05-18T01:04:05.442300+00:00"},{"alias_kind":"arxiv_version","alias_value":"1606.07947v4","created_at":"2026-05-18T01:04:05.442300+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1606.07947","created_at":"2026-05-18T01:04:05.442300+00:00"},{"alias_kind":"pith_short_12","alias_value":"TAVOAQQ7MRAR","created_at":"2026-05-18T12:30:44.179134+00:00"},{"alias_kind":"pith_short_16","alias_value":"TAVOAQQ7MRARLC3L","created_at":"2026-05-18T12:30:44.179134+00:00"},{"alias_kind":"pith_short_8","alias_value":"TAVOAQQ7","created_at":"2026-05-18T12:30:44.179134+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":14,"sample":[{"citing_arxiv_id":"2606.24747","citing_title":"Scaling Laws for Task-Specific LLM Distillation","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30833","citing_title":"Your Teacher Can't Help You Here: Combating Supervision Fidelity Decay in On-Policy Distillation","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"1907.06017","citing_title":"Learn Spelling from Teachers: Transferring Knowledge from Language Models to Sequence-to-Sequence Speech Recognition","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"1907.11804","citing_title":"Memory- and Communication-Aware Model Compression for Distributed Deep Learning Inference on IoT","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2604.14084","citing_title":"TIP: Token Importance in On-Policy Distillation","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.21606","citing_title":"When Are Teacher Tokens Reliable? Position-Weighted On-Policy Self-Distillation for Reasoning","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2604.14084","citing_title":"TIP: Token Importance in On-Policy Distillation","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.16826","citing_title":"Decoupling KL and Trajectories: A Unified Perspective for SFT, DAgger, Offline RL, and OPD in LLM Distillation","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2502.10248","citing_title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","ref_index":184,"is_internal_anchor":true},{"citing_arxiv_id":"2408.00724","citing_title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","ref_index":255,"is_internal_anchor":true},{"citing_arxiv_id":"2211.15089","citing_title":"Continuous diffusion for categorical data","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2402.13116","citing_title":"A Survey on Knowledge Distillation of Large Language Models","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2602.20816","citing_title":"Don't Ignore the Tail: Decoupling top-K Probabilities for Efficient Language Model Distillation","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2603.11178","citing_title":"PACED: Distillation and On-Policy Self-Distillation at the Frontier of Student Competence","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.05940","citing_title":"Near-Policy: Accelerating On-Policy Distillation via Asynchronous Generation and Selective Packing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2302.01318","citing_title":"Accelerating Large Language Model Decoding with Speculative Sampling","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14084","citing_title":"TIP: Token Importance in On-Policy Distillation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM","json":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM.json","graph_json":"https://pith.science/api/pith-number/TAVOAQQ7MRARLC3LHHIHCGIYKM/graph.json","events_json":"https://pith.science/api/pith-number/TAVOAQQ7MRARLC3LHHIHCGIYKM/events.json","paper":"https://pith.science/paper/TAVOAQQ7"},"agent_actions":{"view_html":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM","download_json":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM.json","view_paper":"https://pith.science/paper/TAVOAQQ7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1606.07947&json=true","fetch_graph":"https://pith.science/api/pith-number/TAVOAQQ7MRARLC3LHHIHCGIYKM/graph.json","fetch_events":"https://pith.science/api/pith-number/TAVOAQQ7MRARLC3LHHIHCGIYKM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM/action/storage_attestation","attest_author":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM/action/author_attestation","sign_citation":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM/action/citation_signature","submit_replication":"https://pith.science/pith/TAVOAQQ7MRARLC3LHHIHCGIYKM/action/replication_record"}},"created_at":"2026-05-18T01:04:05.442300+00:00","updated_at":"2026-05-18T01:04:05.442300+00:00"}