{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PFL7JURCDDIKQHZT6NI4V2HESU","short_pith_number":"pith:PFL7JURC","schema_version":"1.0","canonical_sha256":"7957f4d22218d0a81f33f351cae8e4951f7c9555e9d1a2d1d542a7b79a9e79c3","source":{"kind":"arxiv","id":"2310.07177","version":4},"attestation_state":"computed","paper":{"title":"Online Speculative Decoding","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alvin Cheung, Hao Zhang, Ion Stoica, Lanxiang Hu, Peter Bailis, Xiaoxuan Liu, Zhijie Deng","submitted_at":"2023-10-11T04:03:42Z","abstract_excerpt":"Speculative decoding is a pivotal technique to accelerate the inference of large language models (LLMs) by employing a smaller draft model to predict the target model's outputs. However, its efficacy can be limited due to the low predictive accuracy of the draft model, particularly when faced with diverse text inputs and a significant capability gap between the draft and target models. We introduce online speculative decoding to address this challenge. The main idea is to continuously update the (multiple) draft model(s) on observed user query data. Adapting to query distribution mitigates the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07177","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.AI","submitted_at":"2023-10-11T04:03:42Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"621b370fe362554f4549fc7309e08097c43913e1f8b2b7b5d8d18903ac850a1b","abstract_canon_sha256":"b49ad79212ebc1eaeb2f9218bf1989d84383dd25cde84e9dbe933c41b2e001bb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:17.961367Z","signature_b64":"nONT4TD0viTqaDMl6G6U2H6Yfl6QkCkqO0+xcaHr1f1wcio0kLYEAqKkzqHkAmE1gpCoPlilebX5vbHsjunSBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7957f4d22218d0a81f33f351cae8e4951f7c9555e9d1a2d1d542a7b79a9e79c3","last_reissued_at":"2026-07-05T08:29:17.960906Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:17.960906Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Online Speculative Decoding","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Alvin Cheung, Hao Zhang, Ion Stoica, Lanxiang Hu, Peter Bailis, Xiaoxuan Liu, Zhijie Deng","submitted_at":"2023-10-11T04:03:42Z","abstract_excerpt":"Speculative decoding is a pivotal technique to accelerate the inference of large language models (LLMs) by employing a smaller draft model to predict the target model's outputs. However, its efficacy can be limited due to the low predictive accuracy of the draft model, particularly when faced with diverse text inputs and a significant capability gap between the draft and target models. We introduce online speculative decoding to address this challenge. The main idea is to continuously update the (multiple) draft model(s) on observed user query data. Adapting to query distribution mitigates the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07177","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07177","created_at":"2026-07-05T08:29:17.960963+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07177v4","created_at":"2026-07-05T08:29:17.960963+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07177","created_at":"2026-07-05T08:29:17.960963+00:00"},{"alias_kind":"pith_short_12","alias_value":"PFL7JURCDDIK","created_at":"2026-07-05T08:29:17.960963+00:00"},{"alias_kind":"pith_short_16","alias_value":"PFL7JURCDDIKQHZT","created_at":"2026-07-05T08:29:17.960963+00:00"},{"alias_kind":"pith_short_8","alias_value":"PFL7JURC","created_at":"2026-07-05T08:29:17.960963+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04446","citing_title":"D^2SD: Accelerating Speculative Decoding with Dual Diffusion Draft Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02091","citing_title":"DFlare: Scaling Up Draft Capacity for Block Diffusion Speculative Decoding","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29343","citing_title":"Draft-OPD: On-Policy Distillation for Speculative Draft Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00144","citing_title":"BudgetDraft: Acceptance-Aware Multi-View Training for Sparse-KV Speculative Decoding","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09329","citing_title":"Test-Time Speculation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14978","citing_title":"Performance-Driven Policy Optimization for Speculative Decoding with Adaptive Windowing","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06932","citing_title":"When RL Meets Adaptive Speculative Training: A Unified Training-Serving System","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09603","citing_title":"ECHO: Elastic Speculative Decoding with Sparse Gating for High-Concurrency Scenarios","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":240,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15077","citing_title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2412.21187","citing_title":"Do NOT Think That Much for 2+3=? On the Overthinking of o1-Like LLMs","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09329","citing_title":"Test-Time Speculation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05417","citing_title":"Multi-Drafter Speculative Decoding with Alignment Feedback","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21072","citing_title":"Distributed Generative Inference of LLM at Internet Scales with Multi-Dimensional Communication Optimization","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU","json":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU.json","graph_json":"https://pith.science/api/pith-number/PFL7JURCDDIKQHZT6NI4V2HESU/graph.json","events_json":"https://pith.science/api/pith-number/PFL7JURCDDIKQHZT6NI4V2HESU/events.json","paper":"https://pith.science/paper/PFL7JURC"},"agent_actions":{"view_html":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU","download_json":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU.json","view_paper":"https://pith.science/paper/PFL7JURC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07177&json=true","fetch_graph":"https://pith.science/api/pith-number/PFL7JURCDDIKQHZT6NI4V2HESU/graph.json","fetch_events":"https://pith.science/api/pith-number/PFL7JURCDDIKQHZT6NI4V2HESU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU/action/storage_attestation","attest_author":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU/action/author_attestation","sign_citation":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU/action/citation_signature","submit_replication":"https://pith.science/pith/PFL7JURCDDIKQHZT6NI4V2HESU/action/replication_record"}},"created_at":"2026-07-05T08:29:17.960963+00:00","updated_at":"2026-07-05T08:29:17.960963+00:00"}