{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:6RQC4LPENNTOB2LCL6IIYPKAAQ","short_pith_number":"pith:6RQC4LPE","schema_version":"1.0","canonical_sha256":"f4602e2de46b66e0e9625f908c3d400402068e6a915ab77056870d3b833197c4","source":{"kind":"arxiv","id":"2010.01195","version":1},"attestation_state":"computed","paper":{"title":"Leveraging Semantic and Lexical Matching to Improve the Recall of Document Retrieval Systems: A Hybrid Approach","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Cheng Li, Marc Najork, Michael Bendersky, Mingyang Zhang, Saar Kuzi","submitted_at":"2020-10-02T20:59:14Z","abstract_excerpt":"Search engines often follow a two-phase paradigm where in the first stage (the retrieval stage) an initial set of documents is retrieved and in the second stage (the re-ranking stage) the documents are re-ranked to obtain the final result list. While deep neural networks were shown to improve the performance of the re-ranking stage in previous works, there is little literature about using deep neural networks to improve the retrieval stage. In this paper, we study the merits of combining deep neural network models and lexical models for the retrieval stage. A hybrid approach, which leverages b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.01195","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2020-10-02T20:59:14Z","cross_cats_sorted":[],"title_canon_sha256":"7fae59cfe77392e2695dccb85e2082ee9e4789753167571ea0df553aa7416e67","abstract_canon_sha256":"1683da69370cdca714db2072dd156d49f92545f7b9713e6e2bfd13e69ad12edf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:40:02.912037Z","signature_b64":"2m52xyk/0JHLT4GbV/Nq+MZhJKTZY54vhejEekefWAY6EKu7UvlNkbDU4bxNTXH7pPivOt8UeOPBHqUoANp2DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4602e2de46b66e0e9625f908c3d400402068e6a915ab77056870d3b833197c4","last_reissued_at":"2026-07-05T01:40:02.911584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:40:02.911584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Semantic and Lexical Matching to Improve the Recall of Document Retrieval Systems: A Hybrid Approach","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Cheng Li, Marc Najork, Michael Bendersky, Mingyang Zhang, Saar Kuzi","submitted_at":"2020-10-02T20:59:14Z","abstract_excerpt":"Search engines often follow a two-phase paradigm where in the first stage (the retrieval stage) an initial set of documents is retrieved and in the second stage (the re-ranking stage) the documents are re-ranked to obtain the final result list. While deep neural networks were shown to improve the performance of the re-ranking stage in previous works, there is little literature about using deep neural networks to improve the retrieval stage. In this paper, we study the merits of combining deep neural network models and lexical models for the retrieval stage. A hybrid approach, which leverages b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.01195","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.01195/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.01195","created_at":"2026-07-05T01:40:02.911642+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.01195v1","created_at":"2026-07-05T01:40:02.911642+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.01195","created_at":"2026-07-05T01:40:02.911642+00:00"},{"alias_kind":"pith_short_12","alias_value":"6RQC4LPENNTO","created_at":"2026-07-05T01:40:02.911642+00:00"},{"alias_kind":"pith_short_16","alias_value":"6RQC4LPENNTOB2LC","created_at":"2026-07-05T01:40:02.911642+00:00"},{"alias_kind":"pith_short_8","alias_value":"6RQC4LPE","created_at":"2026-07-05T01:40:02.911642+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.05038","citing_title":"Guided Query Refinement: Multimodal Hybrid Retrieval with Test-Time Optimization","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08549","citing_title":"VerifAI: A Verifiable Open-Source Search Engine for Biomedical Question Answering","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23396","citing_title":"Lost in Decoding? Reproducing and Stress-Testing the Look-Ahead Prior in Generative Retrieval","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ","json":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ.json","graph_json":"https://pith.science/api/pith-number/6RQC4LPENNTOB2LCL6IIYPKAAQ/graph.json","events_json":"https://pith.science/api/pith-number/6RQC4LPENNTOB2LCL6IIYPKAAQ/events.json","paper":"https://pith.science/paper/6RQC4LPE"},"agent_actions":{"view_html":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ","download_json":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ.json","view_paper":"https://pith.science/paper/6RQC4LPE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.01195&json=true","fetch_graph":"https://pith.science/api/pith-number/6RQC4LPENNTOB2LCL6IIYPKAAQ/graph.json","fetch_events":"https://pith.science/api/pith-number/6RQC4LPENNTOB2LCL6IIYPKAAQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ/action/storage_attestation","attest_author":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ/action/author_attestation","sign_citation":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ/action/citation_signature","submit_replication":"https://pith.science/pith/6RQC4LPENNTOB2LCL6IIYPKAAQ/action/replication_record"}},"created_at":"2026-07-05T01:40:02.911642+00:00","updated_at":"2026-07-05T01:40:02.911642+00:00"}