{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UVGNT6JVOP73XAQTN23SW4TNKS","short_pith_number":"pith:UVGNT6JV","schema_version":"1.0","canonical_sha256":"a54cd9f93573ffbb82136eb72b726d548db5c3650e0147069c7162e365001ec0","source":{"kind":"arxiv","id":"2310.08461","version":2},"attestation_state":"computed","paper":{"title":"DistillSpec: Improving Speculative Decoding via Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aditya Krishna Menon, Afshin Rostamizadeh, Ankit Singh Rawat, Jean-Fran\\c{c}ois Kagy, Kaifeng Lyu, Rishabh Agarwal, Sanjiv Kumar, Yongchao Zhou","submitted_at":"2023-10-12T16:21:04Z","abstract_excerpt":"Speculative decoding (SD) accelerates large language model inference by employing a faster draft model for generating multiple tokens, which are then verified in parallel by the larger target model, resulting in the text generated according to the target model distribution. However, identifying a compact draft model that is well-aligned with the target model is challenging. To tackle this issue, we propose DistillSpec that uses knowledge distillation to better align the draft model with the target model, before applying SD. DistillSpec makes two key design choices, which we demonstrate via sys"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.08461","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-12T16:21:04Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"bfc48a41d9d0ecfffbbbdc9398d7d1959e43eece72938b4bba32207c8b24dfee","abstract_canon_sha256":"e3584515741f8969f1444a88ed03375cd317460f5441bf237ffffe959faef667"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:22.157750Z","signature_b64":"l9uOFx2BfNxA0SzPlvGSCJTZxRuB9H9BjPmucJXNR8EttrbNctGQXTRWBVpBnz9FPo5kAR2QEkXjLhSzPe95AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a54cd9f93573ffbb82136eb72b726d548db5c3650e0147069c7162e365001ec0","last_reissued_at":"2026-07-05T08:02:22.157225Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:22.157225Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DistillSpec: Improving Speculative Decoding via Knowledge Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aditya Krishna Menon, Afshin Rostamizadeh, Ankit Singh Rawat, Jean-Fran\\c{c}ois Kagy, Kaifeng Lyu, Rishabh Agarwal, Sanjiv Kumar, Yongchao Zhou","submitted_at":"2023-10-12T16:21:04Z","abstract_excerpt":"Speculative decoding (SD) accelerates large language model inference by employing a faster draft model for generating multiple tokens, which are then verified in parallel by the larger target model, resulting in the text generated according to the target model distribution. However, identifying a compact draft model that is well-aligned with the target model is challenging. To tackle this issue, we propose DistillSpec that uses knowledge distillation to better align the draft model with the target model, before applying SD. DistillSpec makes two key design choices, which we demonstrate via sys"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.08461","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.08461/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.08461","created_at":"2026-07-05T08:02:22.157287+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.08461v2","created_at":"2026-07-05T08:02:22.157287+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.08461","created_at":"2026-07-05T08:02:22.157287+00:00"},{"alias_kind":"pith_short_12","alias_value":"UVGNT6JVOP73","created_at":"2026-07-05T08:02:22.157287+00:00"},{"alias_kind":"pith_short_16","alias_value":"UVGNT6JVOP73XAQT","created_at":"2026-07-05T08:02:22.157287+00:00"},{"alias_kind":"pith_short_8","alias_value":"UVGNT6JV","created_at":"2026-07-05T08:02:22.157287+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05394","citing_title":"Weak-to-Strong Generalization via Direct On-Policy Distillation","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06078","citing_title":"Knowledge Distillation for Visual Autoregressive Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02091","citing_title":"DFlare: Scaling Up Draft Capacity for Block Diffusion Speculative Decoding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29343","citing_title":"Draft-OPD: On-Policy Distillation for Speculative Draft Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00626","citing_title":"A Survey of On-Policy Distillation for Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14978","citing_title":"Performance-Driven Policy Optimization for Speculative Decoding with Adaptive Windowing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06019","citing_title":"Multi-Token Prediction via Self-Distillation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09603","citing_title":"ECHO: Elastic Speculative Decoding with Sparse Gating for High-Concurrency Scenarios","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":238,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15077","citing_title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00626","citing_title":"A Survey of On-Policy Distillation for Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27747","citing_title":"Position-Aware Drafting for Inference Acceleration in LLM-Based Generative List-Wise Recommendation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08632","citing_title":"PARD-2: Target-Aligned Parallel Draft Model for Dual-Mode Speculative Decoding","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14682","citing_title":"Acceptance Dynamics Across Cognitive Domains in Speculative Decoding","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS","json":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS.json","graph_json":"https://pith.science/api/pith-number/UVGNT6JVOP73XAQTN23SW4TNKS/graph.json","events_json":"https://pith.science/api/pith-number/UVGNT6JVOP73XAQTN23SW4TNKS/events.json","paper":"https://pith.science/paper/UVGNT6JV"},"agent_actions":{"view_html":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS","download_json":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS.json","view_paper":"https://pith.science/paper/UVGNT6JV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.08461&json=true","fetch_graph":"https://pith.science/api/pith-number/UVGNT6JVOP73XAQTN23SW4TNKS/graph.json","fetch_events":"https://pith.science/api/pith-number/UVGNT6JVOP73XAQTN23SW4TNKS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS/action/storage_attestation","attest_author":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS/action/author_attestation","sign_citation":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS/action/citation_signature","submit_replication":"https://pith.science/pith/UVGNT6JVOP73XAQTN23SW4TNKS/action/replication_record"}},"created_at":"2026-07-05T08:02:22.157287+00:00","updated_at":"2026-07-05T08:02:22.157287+00:00"}