{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G7JHABCVAFKYQ7HK5H7WNL5L7D","short_pith_number":"pith:G7JHABCV","schema_version":"1.0","canonical_sha256":"37d27004550155887ceae9ff66afabf8c7f75d38eec4db843ce90e179218686e","source":{"kind":"arxiv","id":"2404.07220","version":2},"attestation_state":"computed","paper":{"title":"Blended RAG: Improving RAG (Retriever-Augmented Generation) Accuracy with Semantic Search and Hybrid Query-Based Retrievers","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.IR","authors_text":"Abhilasha Mangal, Kunal Sawarkar, Shivam Raj Solanki","submitted_at":"2024-03-22T17:13:46Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) is a prevalent approach to infuse a private knowledge base of documents with Large Language Models (LLM) to build Generative Q\\&A (Question-Answering) systems. However, RAG accuracy becomes increasingly challenging as the corpus of documents scales up, with Retrievers playing an outsized role in the overall RAG accuracy by extracting the most relevant document from the corpus to provide context to the LLM. In this paper, we propose the 'Blended RAG' method of leveraging semantic search techniques, such as Dense Vector indexes and Sparse Encoder indexes, ble"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.07220","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.IR","submitted_at":"2024-03-22T17:13:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"30995c6adc343089e969c05ede53f9febc23d265cd1376f6690928dc68bb366e","abstract_canon_sha256":"7019edc00991506494d904578b3c096d86f629b6df5167eec30494ca0c2aa27c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:02.308736Z","signature_b64":"Updko2+EJJSHPznXq5qf81lOZh7sqf1swuWl3gcedelbx33IKqXYHrkCWe2mkz9LfHhbLP21gVl2yCMy7PfxBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37d27004550155887ceae9ff66afabf8c7f75d38eec4db843ce90e179218686e","last_reissued_at":"2026-07-05T10:26:02.308180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:02.308180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Blended RAG: Improving RAG (Retriever-Augmented Generation) Accuracy with Semantic Search and Hybrid Query-Based Retrievers","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.IR","authors_text":"Abhilasha Mangal, Kunal Sawarkar, Shivam Raj Solanki","submitted_at":"2024-03-22T17:13:46Z","abstract_excerpt":"Retrieval-Augmented Generation (RAG) is a prevalent approach to infuse a private knowledge base of documents with Large Language Models (LLM) to build Generative Q\\&A (Question-Answering) systems. However, RAG accuracy becomes increasingly challenging as the corpus of documents scales up, with Retrievers playing an outsized role in the overall RAG accuracy by extracting the most relevant document from the corpus to provide context to the LLM. In this paper, we propose the 'Blended RAG' method of leveraging semantic search techniques, such as Dense Vector indexes and Sparse Encoder indexes, ble"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.07220","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.07220/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.07220","created_at":"2026-07-05T10:26:02.308255+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.07220v2","created_at":"2026-07-05T10:26:02.308255+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.07220","created_at":"2026-07-05T10:26:02.308255+00:00"},{"alias_kind":"pith_short_12","alias_value":"G7JHABCVAFKY","created_at":"2026-07-05T10:26:02.308255+00:00"},{"alias_kind":"pith_short_16","alias_value":"G7JHABCVAFKYQ7HK","created_at":"2026-07-05T10:26:02.308255+00:00"},{"alias_kind":"pith_short_8","alias_value":"G7JHABCV","created_at":"2026-07-05T10:26:02.308255+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01481","citing_title":"TSGuard: Automated User-Centric Incident Diagnosis for AI Workloads in the Cloud","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2506.18027","citing_title":"PDF Retrieval Augmented Question Answering","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19473","citing_title":"Retrieval-Augmented Generation for AI-Generated Content: A Survey","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05287","citing_title":"Securing the Agent: Vendor-Neutral, Multitenant Enterprise Retrieval and Tool Use","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D","json":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D.json","graph_json":"https://pith.science/api/pith-number/G7JHABCVAFKYQ7HK5H7WNL5L7D/graph.json","events_json":"https://pith.science/api/pith-number/G7JHABCVAFKYQ7HK5H7WNL5L7D/events.json","paper":"https://pith.science/paper/G7JHABCV"},"agent_actions":{"view_html":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D","download_json":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D.json","view_paper":"https://pith.science/paper/G7JHABCV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.07220&json=true","fetch_graph":"https://pith.science/api/pith-number/G7JHABCVAFKYQ7HK5H7WNL5L7D/graph.json","fetch_events":"https://pith.science/api/pith-number/G7JHABCVAFKYQ7HK5H7WNL5L7D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D/action/storage_attestation","attest_author":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D/action/author_attestation","sign_citation":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D/action/citation_signature","submit_replication":"https://pith.science/pith/G7JHABCVAFKYQ7HK5H7WNL5L7D/action/replication_record"}},"created_at":"2026-07-05T10:26:02.308255+00:00","updated_at":"2026-07-05T10:26:02.308255+00:00"}