{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PBA2UZ46SVWWIT7N7CJAPQ5FE3","short_pith_number":"pith:PBA2UZ46","schema_version":"1.0","canonical_sha256":"7841aa679e956d644fedf89207c3a526d02b501c848cd72af810fc06a942061d","source":{"kind":"arxiv","id":"2311.08252","version":2},"attestation_state":"computed","paper":{"title":"REST: Retrieval-Based Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Di He, Jason D. Lee, Tianle Cai, Zexuan Zhong, Zhenyu He","submitted_at":"2023-11-14T15:43:47Z","abstract_excerpt":"We introduce Retrieval-Based Speculative Decoding (REST), a novel algorithm designed to speed up language model generation. The key insight driving the development of REST is the observation that the process of text generation often includes certain common phases and patterns. Unlike previous methods that rely on a draft language model for speculative decoding, REST harnesses the power of retrieval to generate draft tokens. This method draws from the reservoir of existing knowledge, retrieving and employing relevant tokens based on the current context. Its plug-and-play nature allows for seaml"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.08252","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-14T15:43:47Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG"],"title_canon_sha256":"9fa1343dc3777a01c51b6e3df6f1928268cb95a2eddffe79f34778df054492e9","abstract_canon_sha256":"fb93c8dcdc2b5b871c816a935347d91f592211443177b7a4e435e7ea6a184929"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:04:15.903596Z","signature_b64":"C4LMJdhKI9PFgFClpc5sC6O72lYP/7zp4nyFym6eoXUBKapOK8gasAQkSJOJ4wt+9VeV+lAXZaYyDDm/rWwWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7841aa679e956d644fedf89207c3a526d02b501c848cd72af810fc06a942061d","last_reissued_at":"2026-07-05T08:04:15.903179Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:04:15.903179Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"REST: Retrieval-Based Speculative Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Di He, Jason D. Lee, Tianle Cai, Zexuan Zhong, Zhenyu He","submitted_at":"2023-11-14T15:43:47Z","abstract_excerpt":"We introduce Retrieval-Based Speculative Decoding (REST), a novel algorithm designed to speed up language model generation. The key insight driving the development of REST is the observation that the process of text generation often includes certain common phases and patterns. Unlike previous methods that rely on a draft language model for speculative decoding, REST harnesses the power of retrieval to generate draft tokens. This method draws from the reservoir of existing knowledge, retrieving and employing relevant tokens based on the current context. Its plug-and-play nature allows for seaml"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.08252","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.08252/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.08252","created_at":"2026-07-05T08:04:15.903243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.08252v2","created_at":"2026-07-05T08:04:15.903243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.08252","created_at":"2026-07-05T08:04:15.903243+00:00"},{"alias_kind":"pith_short_12","alias_value":"PBA2UZ46SVWW","created_at":"2026-07-05T08:04:15.903243+00:00"},{"alias_kind":"pith_short_16","alias_value":"PBA2UZ46SVWWIT7N","created_at":"2026-07-05T08:04:15.903243+00:00"},{"alias_kind":"pith_short_8","alias_value":"PBA2UZ46","created_at":"2026-07-05T08:04:15.903243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25097","citing_title":"Speculative Decoding at Temperature Zero: A Scoped Safety-Invariance Screen with a 48,072-Sample Expansion","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19473","citing_title":"Retrieval-Augmented Generation for AI-Generated Content: A Survey","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":242,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15077","citing_title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3","json":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3.json","graph_json":"https://pith.science/api/pith-number/PBA2UZ46SVWWIT7N7CJAPQ5FE3/graph.json","events_json":"https://pith.science/api/pith-number/PBA2UZ46SVWWIT7N7CJAPQ5FE3/events.json","paper":"https://pith.science/paper/PBA2UZ46"},"agent_actions":{"view_html":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3","download_json":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3.json","view_paper":"https://pith.science/paper/PBA2UZ46","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.08252&json=true","fetch_graph":"https://pith.science/api/pith-number/PBA2UZ46SVWWIT7N7CJAPQ5FE3/graph.json","fetch_events":"https://pith.science/api/pith-number/PBA2UZ46SVWWIT7N7CJAPQ5FE3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3/action/storage_attestation","attest_author":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3/action/author_attestation","sign_citation":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3/action/citation_signature","submit_replication":"https://pith.science/pith/PBA2UZ46SVWWIT7N7CJAPQ5FE3/action/replication_record"}},"created_at":"2026-07-05T08:04:15.903243+00:00","updated_at":"2026-07-05T08:04:15.903243+00:00"}