{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:6ME6MMIYGGT5WM6A5DGWBZAW2E","short_pith_number":"pith:6ME6MMIY","schema_version":"1.0","canonical_sha256":"f309e6311831a7db33c0e8cd60e416d117a65ffd30f82830e271d6a3303f2b5a","source":{"kind":"arxiv","id":"2602.13836","version":2},"attestation_state":"computed","paper":{"title":"Speculative Decoding with a Speculative Vocabulary","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandros Kouris, Miles Williams, Rui Li, Stylianos I. Venieris, Young D. Kwon","submitted_at":"2026-02-14T16:10:00Z","abstract_excerpt":"Speculative decoding has rapidly emerged as a leading approach for accelerating language model (LM) inference, as it offers substantial speedups while yielding identical outputs. This relies upon a small draft model, tasked with predicting the outputs of the target model. State-of-the-art speculative decoding methods use a draft model comprising a single decoder layer and output embedding matrix, with the latter dominating drafting time for the latest LMs. Recent work has sought to address this output distribution bottleneck by reducing the vocabulary of the draft model. While this can improve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.13836","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-02-14T16:10:00Z","cross_cats_sorted":[],"title_canon_sha256":"27ada5c2b3bcfce9d03265562627b51bf44d16a2130d600a1c7b9849a8c597e9","abstract_canon_sha256":"d8f1da820473dbd296edbacedf7d80b64af4bffee321c9b9f0616d702bdc32b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-20T02:19:17.866201Z","signature_b64":"EMjpem8ixaaOFgdcwBSc9c5zr2LGheKpXVH84na3uIJflUpJSzj4nA2SS+XxpgB+dyJbY5O0SL0ZSVf26ndjAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f309e6311831a7db33c0e8cd60e416d117a65ffd30f82830e271d6a3303f2b5a","last_reissued_at":"2026-07-20T02:19:17.865196Z","signature_status":"signed_v1","first_computed_at":"2026-07-20T02:19:17.865196Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speculative Decoding with a Speculative Vocabulary","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexandros Kouris, Miles Williams, Rui Li, Stylianos I. Venieris, Young D. Kwon","submitted_at":"2026-02-14T16:10:00Z","abstract_excerpt":"Speculative decoding has rapidly emerged as a leading approach for accelerating language model (LM) inference, as it offers substantial speedups while yielding identical outputs. This relies upon a small draft model, tasked with predicting the outputs of the target model. State-of-the-art speculative decoding methods use a draft model comprising a single decoder layer and output embedding matrix, with the latter dominating drafting time for the latest LMs. Recent work has sought to address this output distribution bottleneck by reducing the vocabulary of the draft model. While this can improve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.13836","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.13836/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.13836","created_at":"2026-07-20T02:19:17.865673+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.13836v2","created_at":"2026-07-20T02:19:17.865673+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.13836","created_at":"2026-07-20T02:19:17.865673+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ME6MMIYGGT5","created_at":"2026-07-20T02:19:17.865673+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ME6MMIYGGT5WM6A","created_at":"2026-07-20T02:19:17.865673+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ME6MMIY","created_at":"2026-07-20T02:19:17.865673+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2607.06763","citing_title":"Trees from Marginals: Autoregressive drafting with factorized priors","ref_index":64,"is_internal_anchor":true},{"citing_arxiv_id":"2605.29707","citing_title":"Domino: Decoupling Causal Modeling from Autoregressive Drafting in Speculative Decoding","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10453","citing_title":"SlimSpec: Low-Rank Draft LM-Head for Accelerated Speculative Decoding","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E","json":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E.json","graph_json":"https://pith.science/api/pith-number/6ME6MMIYGGT5WM6A5DGWBZAW2E/graph.json","events_json":"https://pith.science/api/pith-number/6ME6MMIYGGT5WM6A5DGWBZAW2E/events.json","paper":"https://pith.science/paper/6ME6MMIY"},"agent_actions":{"view_html":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E","download_json":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E.json","view_paper":"https://pith.science/paper/6ME6MMIY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.13836&json=true","fetch_graph":"https://pith.science/api/pith-number/6ME6MMIYGGT5WM6A5DGWBZAW2E/graph.json","fetch_events":"https://pith.science/api/pith-number/6ME6MMIYGGT5WM6A5DGWBZAW2E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E/action/storage_attestation","attest_author":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E/action/author_attestation","sign_citation":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E/action/citation_signature","submit_replication":"https://pith.science/pith/6ME6MMIYGGT5WM6A5DGWBZAW2E/action/replication_record"}},"created_at":"2026-07-20T02:19:17.865673+00:00","updated_at":"2026-07-20T02:19:17.865673+00:00"}