{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4R36P3X6HIK6LGQBT4K3OXOALG","short_pith_number":"pith:4R36P3X6","schema_version":"1.0","canonical_sha256":"e477e7eefe3a15e59a019f15b75dc059aa692dcab3aa419827bc0a816bd5eb5c","source":{"kind":"arxiv","id":"2408.11850","version":3},"attestation_state":"computed","paper":{"title":"PEARL: Parallel Speculative Decoding with Adaptive Draft Length","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jianchen Zhu, Kai Liu, Qitan Lv, Tianyu Liu, Winston Hu, Xiao Sun, Yun Li","submitted_at":"2024-08-13T08:32:06Z","abstract_excerpt":"Speculative decoding (SD), where an extra draft model is employed to provide multiple draft tokens first, and then the original target model verifies these tokens in parallel, has shown great power for LLM inference acceleration. However, existing SD methods suffer from the mutual waiting problem, i.e., the target model gets stuck when the draft model is guessing tokens, and vice versa. This problem is directly incurred by the asynchronous execution of the draft model and the target model and is exacerbated due to the fixed draft length in speculative decoding. To address these challenges, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11850","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-13T08:32:06Z","cross_cats_sorted":[],"title_canon_sha256":"4160ab49f8d9aba01caa53a9c34900ba7bcfb36cf334cc88d2105a3ee448955b","abstract_canon_sha256":"d0b46d0bc8adff5b2004d30a422316303a0e5dc6da42b2a79032ee71afcad46a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:18.964853Z","signature_b64":"JBUcOdYJbLPLN76SJ6plypUOANsJ2oQKdekyqebW0QCti1Lxp0hbhlgI1P+c078CNPoSvDMb18T8/2fZEIbgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e477e7eefe3a15e59a019f15b75dc059aa692dcab3aa419827bc0a816bd5eb5c","last_reissued_at":"2026-07-05T10:15:18.964352Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:18.964352Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PEARL: Parallel Speculative Decoding with Adaptive Draft Length","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jianchen Zhu, Kai Liu, Qitan Lv, Tianyu Liu, Winston Hu, Xiao Sun, Yun Li","submitted_at":"2024-08-13T08:32:06Z","abstract_excerpt":"Speculative decoding (SD), where an extra draft model is employed to provide multiple draft tokens first, and then the original target model verifies these tokens in parallel, has shown great power for LLM inference acceleration. However, existing SD methods suffer from the mutual waiting problem, i.e., the target model gets stuck when the draft model is guessing tokens, and vice versa. This problem is directly incurred by the asynchronous execution of the draft model and the target model and is exacerbated due to the fixed draft length in speculative decoding. To address these challenges, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11850","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11850/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11850","created_at":"2026-07-05T10:15:18.964405+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11850v3","created_at":"2026-07-05T10:15:18.964405+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11850","created_at":"2026-07-05T10:15:18.964405+00:00"},{"alias_kind":"pith_short_12","alias_value":"4R36P3X6HIK6","created_at":"2026-07-05T10:15:18.964405+00:00"},{"alias_kind":"pith_short_16","alias_value":"4R36P3X6HIK6LGQB","created_at":"2026-07-05T10:15:18.964405+00:00"},{"alias_kind":"pith_short_8","alias_value":"4R36P3X6","created_at":"2026-07-05T10:15:18.964405+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12303","citing_title":"From 2D Grids to 1D Tokens: Reforming Shared Representations for Multimodal Image Fusion","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29727","citing_title":"Bastion: Budget-Aware Speculative Decoding with Tree-structured Block Diffusion Drafting","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14978","citing_title":"Performance-Driven Policy Optimization for Speculative Decoding with Adaptive Windowing","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09603","citing_title":"ECHO: Elastic Speculative Decoding with Sparse Gating for High-Concurrency Scenarios","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26412","citing_title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26412","citing_title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09375","citing_title":"31.1 A 14.08-to-135.69Token/s ReRAM-on-Logic Stacked Outlier-Free Large-Language-Model Accelerator with Block-Clustered Weight-Compression and Adaptive Parallel-Speculative-Decoding","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07243","citing_title":"SpecBlock: Block-Iterative Speculative Decoding with Dynamic Tree Drafting","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG","json":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG.json","graph_json":"https://pith.science/api/pith-number/4R36P3X6HIK6LGQBT4K3OXOALG/graph.json","events_json":"https://pith.science/api/pith-number/4R36P3X6HIK6LGQBT4K3OXOALG/events.json","paper":"https://pith.science/paper/4R36P3X6"},"agent_actions":{"view_html":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG","download_json":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG.json","view_paper":"https://pith.science/paper/4R36P3X6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11850&json=true","fetch_graph":"https://pith.science/api/pith-number/4R36P3X6HIK6LGQBT4K3OXOALG/graph.json","fetch_events":"https://pith.science/api/pith-number/4R36P3X6HIK6LGQBT4K3OXOALG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG/action/storage_attestation","attest_author":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG/action/author_attestation","sign_citation":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG/action/citation_signature","submit_replication":"https://pith.science/pith/4R36P3X6HIK6LGQBT4K3OXOALG/action/replication_record"}},"created_at":"2026-07-05T10:15:18.964405+00:00","updated_at":"2026-07-05T10:15:18.964405+00:00"}