{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OVOA424SLRBJET3EYS2KQURYTO","short_pith_number":"pith:OVOA424S","schema_version":"1.0","canonical_sha256":"755c0e6b925c42924f64c4b4a852389b8bc24caec47c98b8dfd05673ef0bda36","source":{"kind":"arxiv","id":"2402.14160","version":2},"attestation_state":"computed","paper":{"title":"Recursive Speculative Decoding: Accelerating LLM Inference via Sampling Without Replacement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christopher Lott, Junyoung Park, Mingu Lee, Mukul Gagrani, Raghavv Goel, Wonseok Jeon","submitted_at":"2024-02-21T22:57:49Z","abstract_excerpt":"Speculative decoding is an inference-acceleration method for large language models (LLMs) where a small language model generates a draft-token sequence which is further verified by the target LLM in parallel. Recent works have advanced this method by establishing a draft-token tree, achieving superior performance over a single-sequence speculative decoding. However, those works independently generate tokens at each level of the tree, not leveraging the tree's entire diversifiability. Besides, their empirical superiority has been shown for fixed length of sequences, implicitly granting more com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14160","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-21T22:57:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e41ceea97bfbf7f02b09d1d715a280e45cd9f974466aeb0d52fe37e4838a7bc2","abstract_canon_sha256":"31f1a4d66a0ff4e7cf095bd0aadafa2d44be221c19b3f2461142b553af64e0c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:52:16.068521Z","signature_b64":"/vdahecxo2fGSn3FiyeLsy42DrYqIowlPopvXK2OE74b4rUITb45dnBn/DrBushhcUZ29xQvlNgwDsAdQ8GuAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"755c0e6b925c42924f64c4b4a852389b8bc24caec47c98b8dfd05673ef0bda36","last_reissued_at":"2026-07-05T07:52:16.068065Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:52:16.068065Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Recursive Speculative Decoding: Accelerating LLM Inference via Sampling Without Replacement","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Christopher Lott, Junyoung Park, Mingu Lee, Mukul Gagrani, Raghavv Goel, Wonseok Jeon","submitted_at":"2024-02-21T22:57:49Z","abstract_excerpt":"Speculative decoding is an inference-acceleration method for large language models (LLMs) where a small language model generates a draft-token sequence which is further verified by the target LLM in parallel. Recent works have advanced this method by establishing a draft-token tree, achieving superior performance over a single-sequence speculative decoding. However, those works independently generate tokens at each level of the tree, not leveraging the tree's entire diversifiability. Besides, their empirical superiority has been shown for fixed length of sequences, implicitly granting more com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14160","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14160/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14160","created_at":"2026-07-05T07:52:16.068123+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14160v2","created_at":"2026-07-05T07:52:16.068123+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14160","created_at":"2026-07-05T07:52:16.068123+00:00"},{"alias_kind":"pith_short_12","alias_value":"OVOA424SLRBJ","created_at":"2026-07-05T07:52:16.068123+00:00"},{"alias_kind":"pith_short_16","alias_value":"OVOA424SLRBJET3E","created_at":"2026-07-05T07:52:16.068123+00:00"},{"alias_kind":"pith_short_8","alias_value":"OVOA424S","created_at":"2026-07-05T07:52:16.068123+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01813","citing_title":"Cost-Aware Diffusion Draft Trees for Speculative Decoding","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08899","citing_title":"ConFu: Contemplate the Future for Better Speculative Sampling","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04987","citing_title":"Cactus: Accelerating Auto-Regressive Decoding with Constrained Acceptance Speculative Sampling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04543","citing_title":"UniVer: A Unified Perspective for Multi-step and Multi-draft Speculative Decoding","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO","json":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO.json","graph_json":"https://pith.science/api/pith-number/OVOA424SLRBJET3EYS2KQURYTO/graph.json","events_json":"https://pith.science/api/pith-number/OVOA424SLRBJET3EYS2KQURYTO/events.json","paper":"https://pith.science/paper/OVOA424S"},"agent_actions":{"view_html":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO","download_json":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO.json","view_paper":"https://pith.science/paper/OVOA424S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14160&json=true","fetch_graph":"https://pith.science/api/pith-number/OVOA424SLRBJET3EYS2KQURYTO/graph.json","fetch_events":"https://pith.science/api/pith-number/OVOA424SLRBJET3EYS2KQURYTO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO/action/storage_attestation","attest_author":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO/action/author_attestation","sign_citation":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO/action/citation_signature","submit_replication":"https://pith.science/pith/OVOA424SLRBJET3EYS2KQURYTO/action/replication_record"}},"created_at":"2026-07-05T07:52:16.068123+00:00","updated_at":"2026-07-05T07:52:16.068123+00:00"}