{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GBRFLG6F42MSH2JXLK2UZHLXUT","short_pith_number":"pith:GBRFLG6F","schema_version":"1.0","canonical_sha256":"3062559bc5e69923e9375ab54c9d77a4ffebc09c8fb370ef1d6a56bc8b55c5b7","source":{"kind":"arxiv","id":"2411.05289","version":1},"attestation_state":"computed","paper":{"title":"SpecHub: Provable Acceleration to Multi-Draft Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Lichao Sun, Ryan Sun, Tianyi Zhou, Xun Chen","submitted_at":"2024-11-08T02:47:07Z","abstract_excerpt":"Large Language Models (LLMs) have become essential in advancing natural language processing (NLP) tasks, but their sequential token generation limits inference speed. Multi-Draft Speculative Decoding (MDSD) offers a promising solution by using a smaller draft model to generate multiple token sequences, which the target LLM verifies in parallel. However, current heuristic approaches, such as Recursive Rejection Sampling (RRS), suffer from low acceptance rates in subsequent drafts, limiting the advantages of using multiple drafts. Meanwhile, Optimal Transport with Membership Cost (OTM) can theor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.05289","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-08T02:47:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"73b5343795a7d088cdedae442ce099340798a3196b6d6b974d55975488b97bfb","abstract_canon_sha256":"6237a3ef32a9526750459898002b6ee777ccd5a4699e5fd3df8e0735d0b46b3b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:57.562143Z","signature_b64":"2EKAx1cQ8tQmdJ1w0wYWoVUMZn7Itr5czcDkCOI4wrcfY0nSI/eW4+RgUYnPQOdo0FtgOldtnQeXNeOCbzWrAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3062559bc5e69923e9375ab54c9d77a4ffebc09c8fb370ef1d6a56bc8b55c5b7","last_reissued_at":"2026-07-05T09:32:57.561634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:57.561634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpecHub: Provable Acceleration to Multi-Draft Speculative Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Lichao Sun, Ryan Sun, Tianyi Zhou, Xun Chen","submitted_at":"2024-11-08T02:47:07Z","abstract_excerpt":"Large Language Models (LLMs) have become essential in advancing natural language processing (NLP) tasks, but their sequential token generation limits inference speed. Multi-Draft Speculative Decoding (MDSD) offers a promising solution by using a smaller draft model to generate multiple token sequences, which the target LLM verifies in parallel. However, current heuristic approaches, such as Recursive Rejection Sampling (RRS), suffer from low acceptance rates in subsequent drafts, limiting the advantages of using multiple drafts. Meanwhile, Optimal Transport with Membership Cost (OTM) can theor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.05289","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.05289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.05289","created_at":"2026-07-05T09:32:57.561696+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.05289v1","created_at":"2026-07-05T09:32:57.561696+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.05289","created_at":"2026-07-05T09:32:57.561696+00:00"},{"alias_kind":"pith_short_12","alias_value":"GBRFLG6F42MS","created_at":"2026-07-05T09:32:57.561696+00:00"},{"alias_kind":"pith_short_16","alias_value":"GBRFLG6F42MSH2JX","created_at":"2026-07-05T09:32:57.561696+00:00"},{"alias_kind":"pith_short_8","alias_value":"GBRFLG6F","created_at":"2026-07-05T09:32:57.561696+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08923","citing_title":"CopySpec: Accelerating LLMs with Speculative Copy-and-Paste Without Compromising Quality","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT","json":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT.json","graph_json":"https://pith.science/api/pith-number/GBRFLG6F42MSH2JXLK2UZHLXUT/graph.json","events_json":"https://pith.science/api/pith-number/GBRFLG6F42MSH2JXLK2UZHLXUT/events.json","paper":"https://pith.science/paper/GBRFLG6F"},"agent_actions":{"view_html":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT","download_json":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT.json","view_paper":"https://pith.science/paper/GBRFLG6F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.05289&json=true","fetch_graph":"https://pith.science/api/pith-number/GBRFLG6F42MSH2JXLK2UZHLXUT/graph.json","fetch_events":"https://pith.science/api/pith-number/GBRFLG6F42MSH2JXLK2UZHLXUT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT/action/storage_attestation","attest_author":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT/action/author_attestation","sign_citation":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT/action/citation_signature","submit_replication":"https://pith.science/pith/GBRFLG6F42MSH2JXLK2UZHLXUT/action/replication_record"}},"created_at":"2026-07-05T09:32:57.561696+00:00","updated_at":"2026-07-05T09:32:57.561696+00:00"}