{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B4LVT5T6A2VJIHNGHCUVJPHITY","short_pith_number":"pith:B4LVT5T6","schema_version":"1.0","canonical_sha256":"0f1759f67e06aa941da638a954bce89e2c479c48027367c8109d37f93e17f3b3","source":{"kind":"arxiv","id":"2402.02057","version":1},"attestation_state":"computed","paper":{"title":"Break the Sequential Dependency of LLM Inference Using Lookahead Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Hao Zhang, Ion Stoica, Peter Bailis, Yichao Fu","submitted_at":"2024-02-03T06:37:50Z","abstract_excerpt":"Autoregressive decoding of large language models (LLMs) is memory bandwidth bounded, resulting in high latency and significant wastes of the parallel processing power of modern accelerators. Existing methods for accelerating LLM decoding often require a draft model (e.g., speculative decoding), which is nontrivial to obtain and unable to generalize. In this paper, we introduce Lookahead decoding, an exact, parallel decoding algorithm that accelerates LLM decoding without needing auxiliary models or data stores. It allows trading per-step log(FLOPs) to reduce the number of total decoding steps,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.02057","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-03T06:37:50Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f5db40cc81e063ac2babfd1d1cd0c3b629f6e0f717b75bd50bfd580da6203c22","abstract_canon_sha256":"9f6c6ab74ab3a9d3282d38df67cc3095d4e1fb5910e9096775070a9fb6cdbf18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:10.323260Z","signature_b64":"TeH8C0MMAe6C6VgFJAiDrmI+3ax8mXRwW5uhOJsAGTlnvzLyEt4Vu5QdTPBnPj4c6SZXD3Xiu1bn1LNy6WpRCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f1759f67e06aa941da638a954bce89e2c479c48027367c8109d37f93e17f3b3","last_reissued_at":"2026-07-05T07:41:10.322810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:10.322810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Break the Sequential Dependency of LLM Inference Using Lookahead Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Hao Zhang, Ion Stoica, Peter Bailis, Yichao Fu","submitted_at":"2024-02-03T06:37:50Z","abstract_excerpt":"Autoregressive decoding of large language models (LLMs) is memory bandwidth bounded, resulting in high latency and significant wastes of the parallel processing power of modern accelerators. Existing methods for accelerating LLM decoding often require a draft model (e.g., speculative decoding), which is nontrivial to obtain and unable to generalize. In this paper, we introduce Lookahead decoding, an exact, parallel decoding algorithm that accelerates LLM decoding without needing auxiliary models or data stores. It allows trading per-step log(FLOPs) to reduce the number of total decoding steps,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.02057","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.02057/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.02057","created_at":"2026-07-05T07:41:10.322863+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.02057v1","created_at":"2026-07-05T07:41:10.322863+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.02057","created_at":"2026-07-05T07:41:10.322863+00:00"},{"alias_kind":"pith_short_12","alias_value":"B4LVT5T6A2VJ","created_at":"2026-07-05T07:41:10.322863+00:00"},{"alias_kind":"pith_short_16","alias_value":"B4LVT5T6A2VJIHNG","created_at":"2026-07-05T07:41:10.322863+00:00"},{"alias_kind":"pith_short_8","alias_value":"B4LVT5T6","created_at":"2026-07-05T07:41:10.322863+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25354","citing_title":"Efficient and Trainable Language Model Test-Time Scaling via Local Branch Routing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25097","citing_title":"Speculative Decoding at Temperature Zero: A Scoped Safety-Invariance Screen with a 48,072-Sample Expansion","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10935","citing_title":"CLP: Collocation-Length Prediction for Zero-Loss Adaptive Multi-Token Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05158","citing_title":"Streaming Communication in Multi-Agent Reasoning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27538","citing_title":"The Context-Ready Transformer","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26558","citing_title":"Cassandra: Enabling Reasoning LLMs at Edge via Self-Speculative Decoding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25698","citing_title":"Reference-Augmented Learning for Precise Tracking Policy of Tendon-Driven Continuum Robots","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25354","citing_title":"Efficient and Trainable Language Model Test-Time Scaling via Local Branch Routing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15051","citing_title":"An Interpretable Latency Model for Speculative Decoding in LLM Serving","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29727","citing_title":"Bastion: Budget-Aware Speculative Decoding with Tree-structured Block Diffusion Drafting","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05465","citing_title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20104","citing_title":"Draft Less, Retrieve More: Hybrid Tree Construction for Speculative Decoding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14978","citing_title":"Performance-Driven Policy Optimization for Speculative Decoding with Adaptive Windowing","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12052","citing_title":"FluentAvatar: Flicker-Free Talking-Head Animation via Phoneme-Guided Autoregressive Modeling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15545","citing_title":"TokenTiming: A Dynamic Alignment Method for Universal Speculative Decoding Model Pairs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13587","citing_title":"VVS: Accelerating Speculative Decoding for Visual Autoregressive Generation via Partial Verification Skipping","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14617","citing_title":"Seer: Online Context Learning for Fast Synchronous LLM Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05524","citing_title":"Double: Breaking the Acceleration Limit via Double Retrieval Speculative Parallelism","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06932","citing_title":"When RL Meets Adaptive Speculative Training: A Unified Training-Serving System","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09603","citing_title":"ECHO: Elastic Speculative Decoding with Sparse Gating for High-Concurrency Scenarios","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11186","citing_title":"CATS: Cascaded Adaptive Tree Speculation for Memory-Limited LLM Inference Acceleration","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25699","citing_title":"NVLLM: A 3D NAND-Centric Architecture Enabling Edge on-Device LLM Inference","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06548","citing_title":"Continuous Latent Diffusion Language Model","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY","json":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY.json","graph_json":"https://pith.science/api/pith-number/B4LVT5T6A2VJIHNGHCUVJPHITY/graph.json","events_json":"https://pith.science/api/pith-number/B4LVT5T6A2VJIHNGHCUVJPHITY/events.json","paper":"https://pith.science/paper/B4LVT5T6"},"agent_actions":{"view_html":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY","download_json":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY.json","view_paper":"https://pith.science/paper/B4LVT5T6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.02057&json=true","fetch_graph":"https://pith.science/api/pith-number/B4LVT5T6A2VJIHNGHCUVJPHITY/graph.json","fetch_events":"https://pith.science/api/pith-number/B4LVT5T6A2VJIHNGHCUVJPHITY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY/action/storage_attestation","attest_author":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY/action/author_attestation","sign_citation":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY/action/citation_signature","submit_replication":"https://pith.science/pith/B4LVT5T6A2VJIHNGHCUVJPHITY/action/replication_record"}},"created_at":"2026-07-05T07:41:10.322863+00:00","updated_at":"2026-07-05T07:41:10.322863+00:00"}