{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AMTO7ISGDCJYZK55HT2EMZABLE","short_pith_number":"pith:AMTO7ISG","schema_version":"1.0","canonical_sha256":"0326efa24618938cabbd3cf44664015912c562b88542ddcc5874b81a60932627","source":{"kind":"arxiv","id":"2407.08223","version":2},"attestation_state":"computed","paper":{"title":"Speculative RAG: Enhancing Retrieval Augmented Generation through Drafting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ankur Taly, Anush Mattapalli, Chen-Yu Lee, Huaixiu Steven Zheng, Jingbo Shang, Long Le, Swaroop Mishra, Tomas Pfister, Vincent Perot, Yuwei Zhang, Zifeng Wang, Zilong Wang","submitted_at":"2024-07-11T06:50:19Z","abstract_excerpt":"Retrieval augmented generation (RAG) combines the generative abilities of large language models (LLMs) with external knowledge sources to provide more accurate and up-to-date responses. Recent RAG advancements focus on improving retrieval outcomes through iterative LLM refinement or self-critique capabilities acquired through additional instruction tuning of LLMs. In this work, we introduce Speculative RAG - a framework that leverages a larger generalist LM to efficiently verify multiple RAG drafts produced in parallel by a smaller, distilled specialist LM. Each draft is generated from a disti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08223","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-11T06:50:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"99fe5297b31678ce782a4bd7488b2d00cbbd79ba441a9792170f236d5b03aa32","abstract_canon_sha256":"bad279453a3e8dce914482808dcd0c45668ceeae04506f400049ff1e6513d51c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:21.634679Z","signature_b64":"C/0RfRjfMTjeB+zaCASwwhHMGg2y131uf4eR2EDZ/b5+tK1NrVx03MsKSUVAaF19EVlXNkERxIJf2oj4mzZ9BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0326efa24618938cabbd3cf44664015912c562b88542ddcc5874b81a60932627","last_reissued_at":"2026-07-05T10:21:21.634071Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:21.634071Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speculative RAG: Enhancing Retrieval Augmented Generation through Drafting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ankur Taly, Anush Mattapalli, Chen-Yu Lee, Huaixiu Steven Zheng, Jingbo Shang, Long Le, Swaroop Mishra, Tomas Pfister, Vincent Perot, Yuwei Zhang, Zifeng Wang, Zilong Wang","submitted_at":"2024-07-11T06:50:19Z","abstract_excerpt":"Retrieval augmented generation (RAG) combines the generative abilities of large language models (LLMs) with external knowledge sources to provide more accurate and up-to-date responses. Recent RAG advancements focus on improving retrieval outcomes through iterative LLM refinement or self-critique capabilities acquired through additional instruction tuning of LLMs. In this work, we introduce Speculative RAG - a framework that leverages a larger generalist LM to efficiently verify multiple RAG drafts produced in parallel by a smaller, distilled specialist LM. Each draft is generated from a disti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08223","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08223/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08223","created_at":"2026-07-05T10:21:21.634132+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08223v2","created_at":"2026-07-05T10:21:21.634132+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08223","created_at":"2026-07-05T10:21:21.634132+00:00"},{"alias_kind":"pith_short_12","alias_value":"AMTO7ISGDCJY","created_at":"2026-07-05T10:21:21.634132+00:00"},{"alias_kind":"pith_short_16","alias_value":"AMTO7ISGDCJYZK55","created_at":"2026-07-05T10:21:21.634132+00:00"},{"alias_kind":"pith_short_8","alias_value":"AMTO7ISG","created_at":"2026-07-05T10:21:21.634132+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17449","citing_title":"MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07235","citing_title":"FLOWREADER: Min-Cost Flow Optimization for Multi-Modal Long Document Q&A","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10177","citing_title":"VArify: A Visual Analytics System for Verifying Knowledge Enhanced Large Language Model Responses in Food Science","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13957","citing_title":"Supervising the search process produces reliable and generalizable information-seeking agents","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21965","citing_title":"SpecHop: Continuous Speculation for Accelerating Multi-Hop Retrieval Agents","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00505","citing_title":"LLM-Oriented Information Retrieval: A Denoising-First Perspective","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18007","citing_title":"Semantic Reranking at Inference Time for Hard Examples in Rhetorical Role Labeling","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00505","citing_title":"LLM-Oriented Information Retrieval: A Denoising-First Perspective","ref_index":193,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE","json":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE.json","graph_json":"https://pith.science/api/pith-number/AMTO7ISGDCJYZK55HT2EMZABLE/graph.json","events_json":"https://pith.science/api/pith-number/AMTO7ISGDCJYZK55HT2EMZABLE/events.json","paper":"https://pith.science/paper/AMTO7ISG"},"agent_actions":{"view_html":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE","download_json":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE.json","view_paper":"https://pith.science/paper/AMTO7ISG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08223&json=true","fetch_graph":"https://pith.science/api/pith-number/AMTO7ISGDCJYZK55HT2EMZABLE/graph.json","fetch_events":"https://pith.science/api/pith-number/AMTO7ISGDCJYZK55HT2EMZABLE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE/action/storage_attestation","attest_author":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE/action/author_attestation","sign_citation":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE/action/citation_signature","submit_replication":"https://pith.science/pith/AMTO7ISGDCJYZK55HT2EMZABLE/action/replication_record"}},"created_at":"2026-07-05T10:21:21.634132+00:00","updated_at":"2026-07-05T10:21:21.634132+00:00"}