{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RQB5FVJO3SXTPJ46S4OON4F5P2","short_pith_number":"pith:RQB5FVJO","schema_version":"1.0","canonical_sha256":"8c03d2d52edcaf37a79e971ce6f0bd7e849a56fcc2187d17fc068276f61ac299","source":{"kind":"arxiv","id":"2401.07851","version":3},"attestation_state":"computed","paper":{"title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Heming Xia, Peiyi Wang, Qingxiu Dong, Tao Ge, Tianyu Liu, Wenjie Li, Yongqi Li, Zhe Yang, Zhifang Sui","submitted_at":"2024-01-15T17:26:50Z","abstract_excerpt":"To mitigate the high inference latency stemming from autoregressive decoding in Large Language Models (LLMs), Speculative Decoding has emerged as a novel decoding paradigm for LLM inference. In each decoding step, this method first drafts several future tokens efficiently and then verifies them in parallel. Unlike autoregressive decoding, Speculative Decoding facilitates the simultaneous decoding of multiple tokens per step, thereby accelerating inference. This paper presents a comprehensive overview and analysis of this promising decoding paradigm. We begin by providing a formal definition an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.07851","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-15T17:26:50Z","cross_cats_sorted":[],"title_canon_sha256":"490567fe85edecc19837b6879c3d123ce9d8953690aee4a0a0dcbe8d4043792b","abstract_canon_sha256":"3a5e156bf2fefe52f78ee0671ec0c76361e96850fee130d787358819f507bfea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:02.390491Z","signature_b64":"I+2YJOcxTO5jDNSDfRgQbmBpbDeQ6qGilokVw5w1+ioGb1dAQUBfjtWvENmmCp8CEN44E4jk6iCEQVOkq71eDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c03d2d52edcaf37a79e971ce6f0bd7e849a56fcc2187d17fc068276f61ac299","last_reissued_at":"2026-07-05T08:27:02.390000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:02.390000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Heming Xia, Peiyi Wang, Qingxiu Dong, Tao Ge, Tianyu Liu, Wenjie Li, Yongqi Li, Zhe Yang, Zhifang Sui","submitted_at":"2024-01-15T17:26:50Z","abstract_excerpt":"To mitigate the high inference latency stemming from autoregressive decoding in Large Language Models (LLMs), Speculative Decoding has emerged as a novel decoding paradigm for LLM inference. In each decoding step, this method first drafts several future tokens efficiently and then verifies them in parallel. Unlike autoregressive decoding, Speculative Decoding facilitates the simultaneous decoding of multiple tokens per step, thereby accelerating inference. This paper presents a comprehensive overview and analysis of this promising decoding paradigm. We begin by providing a formal definition an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.07851","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.07851/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.07851","created_at":"2026-07-05T08:27:02.390064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.07851v3","created_at":"2026-07-05T08:27:02.390064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.07851","created_at":"2026-07-05T08:27:02.390064+00:00"},{"alias_kind":"pith_short_12","alias_value":"RQB5FVJO3SXT","created_at":"2026-07-05T08:27:02.390064+00:00"},{"alias_kind":"pith_short_16","alias_value":"RQB5FVJO3SXTPJ46","created_at":"2026-07-05T08:27:02.390064+00:00"},{"alias_kind":"pith_short_8","alias_value":"RQB5FVJO","created_at":"2026-07-05T08:27:02.390064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25091","citing_title":"Speculation at a Distance: Where Edge-Cloud Speculative Decoding Actually Pays Off","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04921","citing_title":"SURF: Separation via Unsupervised Remixing Flow","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29639","citing_title":"RTP-LLM: High-Performance Alibaba LLM Inference Engine","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03300","citing_title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05465","citing_title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18810","citing_title":"D-PACE: Dynamic Position-Aware Cross-Entropy for Parallel Speculative Drafting","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20104","citing_title":"Draft Less, Retrieve More: Hybrid Tree Construction for Speculative Decoding","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06932","citing_title":"When RL Meets Adaptive Speculative Training: A Unified Training-Serving System","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09603","citing_title":"ECHO: Elastic Speculative Decoding with Sparse Gating for High-Concurrency Scenarios","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10195","citing_title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":269,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26412","citing_title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26412","citing_title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10195","citing_title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22906","citing_title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20503","citing_title":"FASER: Fine-Grained Phase Management for Speculative Decoding in Dynamic LLM Serving","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2","json":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2.json","graph_json":"https://pith.science/api/pith-number/RQB5FVJO3SXTPJ46S4OON4F5P2/graph.json","events_json":"https://pith.science/api/pith-number/RQB5FVJO3SXTPJ46S4OON4F5P2/events.json","paper":"https://pith.science/paper/RQB5FVJO"},"agent_actions":{"view_html":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2","download_json":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2.json","view_paper":"https://pith.science/paper/RQB5FVJO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.07851&json=true","fetch_graph":"https://pith.science/api/pith-number/RQB5FVJO3SXTPJ46S4OON4F5P2/graph.json","fetch_events":"https://pith.science/api/pith-number/RQB5FVJO3SXTPJ46S4OON4F5P2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2/action/storage_attestation","attest_author":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2/action/author_attestation","sign_citation":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2/action/citation_signature","submit_replication":"https://pith.science/pith/RQB5FVJO3SXTPJ46S4OON4F5P2/action/replication_record"}},"created_at":"2026-07-05T08:27:02.390064+00:00","updated_at":"2026-07-05T08:27:02.390064+00:00"}