{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OOZUVZ2HVKHHF6NYU4ZQZWTMCL","short_pith_number":"pith:OOZUVZ2H","schema_version":"1.0","canonical_sha256":"73b34ae747aa8e72f9b8a7330cda6c12c588c4f5523cf861f6453ffc38b67e8e","source":{"kind":"arxiv","id":"2410.18351","version":1},"attestation_state":"computed","paper":{"title":"AdaEDL: Early Draft Stopping for Speculative Decoding of Large Language Models via an Entropy-based Lower Bound on Token Acceptance Probability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Mingu Lee, Sudhanshu Agrawal, Wonseok Jeon","submitted_at":"2024-10-24T01:13:43Z","abstract_excerpt":"Speculative decoding is a powerful technique that attempts to circumvent the autoregressive constraint of modern Large Language Models (LLMs). The aim of speculative decoding techniques is to improve the average inference time of a large, target model without sacrificing its accuracy, by using a more efficient draft model to propose draft tokens which are then verified in parallel. The number of draft tokens produced in each drafting round is referred to as the draft length and is often a static hyperparameter chosen based on the acceptance rate statistics of the draft tokens. However, setting"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18351","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T01:13:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"365cc3c90e1a0e30e06a4134344347349c222f8956d79fcfce70776ba68b692d","abstract_canon_sha256":"99dc4ab12c39e03446a392afafed33e7d1c0d5de2991116c6e506d768689c044"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:07.099845Z","signature_b64":"cOMNsmbMlWRwEaux8i/Hzr8kcItLjW5iKdFndRosRuB+E8DAH0Mh32In1lW5Huhb/I7Egf7ZBCQFWCSQHHkxDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73b34ae747aa8e72f9b8a7330cda6c12c588c4f5523cf861f6453ffc38b67e8e","last_reissued_at":"2026-07-05T09:25:07.099380Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:07.099380Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AdaEDL: Early Draft Stopping for Speculative Decoding of Large Language Models via an Entropy-based Lower Bound on Token Acceptance Probability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Mingu Lee, Sudhanshu Agrawal, Wonseok Jeon","submitted_at":"2024-10-24T01:13:43Z","abstract_excerpt":"Speculative decoding is a powerful technique that attempts to circumvent the autoregressive constraint of modern Large Language Models (LLMs). The aim of speculative decoding techniques is to improve the average inference time of a large, target model without sacrificing its accuracy, by using a more efficient draft model to propose draft tokens which are then verified in parallel. The number of draft tokens produced in each drafting round is referred to as the draft length and is often a static hyperparameter chosen based on the acceptance rate statistics of the draft tokens. However, setting"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18351","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18351/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18351","created_at":"2026-07-05T09:25:07.099444+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18351v1","created_at":"2026-07-05T09:25:07.099444+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18351","created_at":"2026-07-05T09:25:07.099444+00:00"},{"alias_kind":"pith_short_12","alias_value":"OOZUVZ2HVKHH","created_at":"2026-07-05T09:25:07.099444+00:00"},{"alias_kind":"pith_short_16","alias_value":"OOZUVZ2HVKHHF6NY","created_at":"2026-07-05T09:25:07.099444+00:00"},{"alias_kind":"pith_short_8","alias_value":"OOZUVZ2H","created_at":"2026-07-05T09:25:07.099444+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17613","citing_title":"VeriCache: Turning Lossy KV Cache into Lossless LLM Inference","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05524","citing_title":"Double: Breaking the Acceleration Limit via Double Retrieval Speculative Parallelism","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL","json":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL.json","graph_json":"https://pith.science/api/pith-number/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/graph.json","events_json":"https://pith.science/api/pith-number/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/events.json","paper":"https://pith.science/paper/OOZUVZ2H"},"agent_actions":{"view_html":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL","download_json":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL.json","view_paper":"https://pith.science/paper/OOZUVZ2H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18351&json=true","fetch_graph":"https://pith.science/api/pith-number/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/graph.json","fetch_events":"https://pith.science/api/pith-number/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/action/storage_attestation","attest_author":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/action/author_attestation","sign_citation":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/action/citation_signature","submit_replication":"https://pith.science/pith/OOZUVZ2HVKHHF6NYU4ZQZWTMCL/action/replication_record"}},"created_at":"2026-07-05T09:25:07.099444+00:00","updated_at":"2026-07-05T09:25:07.099444+00:00"}