{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:KTPORQJTJEHHM7HQWPMUQHDTFX","short_pith_number":"pith:KTPORQJT","schema_version":"1.0","canonical_sha256":"54dee8c133490e767cf0b3d9481c732de36dad7bd3b4242aa83d65b04f521147","source":{"kind":"arxiv","id":"2010.08191","version":2},"attestation_state":"computed","paper":{"title":"RocketQA: An Optimized Training Approach to Dense Passage Retrieval for Open-Domain Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Daxiang Dong, HaiFeng Wang, Hua Wu, Jing Liu, Kai Liu, Ruiyang Ren, Wayne Xin Zhao, Yingqi Qu, Yuchen Ding","submitted_at":"2020-10-16T06:54:05Z","abstract_excerpt":"In open-domain question answering, dense passage retrieval has become a new paradigm to retrieve relevant passages for finding answers. Typically, the dual-encoder architecture is adopted to learn dense representations of questions and passages for semantic matching. However, it is difficult to effectively train a dual-encoder due to the challenges including the discrepancy between training and inference, the existence of unlabeled positives and limited training data. To address these challenges, we propose an optimized training approach, called RocketQA, to improving dense passage retrieval. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.08191","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-16T06:54:05Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"51f555ab3a26cfd43a977f181e57ba78fb3693d21cda74271046e2fa790d6a50","abstract_canon_sha256":"9de5a2ec1c0275f91251a951504f10116bcf0af6d807fc1fd999ff2814c7911b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:39:42.782718Z","signature_b64":"+LLRmz8w+cc8jynjEA8LdG3dO82b/LOEAKU7WnziQk3BRXlaRfX6VYDaFXxX5F72uPGVNnJT17CSSnvRVFqzBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54dee8c133490e767cf0b3d9481c732de36dad7bd3b4242aa83d65b04f521147","last_reissued_at":"2026-07-05T02:39:42.782290Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:39:42.782290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RocketQA: An Optimized Training Approach to Dense Passage Retrieval for Open-Domain Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Daxiang Dong, HaiFeng Wang, Hua Wu, Jing Liu, Kai Liu, Ruiyang Ren, Wayne Xin Zhao, Yingqi Qu, Yuchen Ding","submitted_at":"2020-10-16T06:54:05Z","abstract_excerpt":"In open-domain question answering, dense passage retrieval has become a new paradigm to retrieve relevant passages for finding answers. Typically, the dual-encoder architecture is adopted to learn dense representations of questions and passages for semantic matching. However, it is difficult to effectively train a dual-encoder due to the challenges including the discrepancy between training and inference, the existence of unlabeled positives and limited training data. To address these challenges, we propose an optimized training approach, called RocketQA, to improving dense passage retrieval. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.08191","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.08191/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.08191","created_at":"2026-07-05T02:39:42.782363+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.08191v2","created_at":"2026-07-05T02:39:42.782363+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.08191","created_at":"2026-07-05T02:39:42.782363+00:00"},{"alias_kind":"pith_short_12","alias_value":"KTPORQJTJEHH","created_at":"2026-07-05T02:39:42.782363+00:00"},{"alias_kind":"pith_short_16","alias_value":"KTPORQJTJEHHM7HQ","created_at":"2026-07-05T02:39:42.782363+00:00"},{"alias_kind":"pith_short_8","alias_value":"KTPORQJT","created_at":"2026-07-05T02:39:42.782363+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16621","citing_title":"The Role of Vocabularies in Learning Sparse Representations for Ranking","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05038","citing_title":"Guided Query Refinement: Multimodal Hybrid Retrieval with Test-Time Optimization","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07597","citing_title":"C-Pack: Packed Resources For General Chinese Embeddings","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03216","citing_title":"M3-Embedding: Multi-Linguality, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX","json":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX.json","graph_json":"https://pith.science/api/pith-number/KTPORQJTJEHHM7HQWPMUQHDTFX/graph.json","events_json":"https://pith.science/api/pith-number/KTPORQJTJEHHM7HQWPMUQHDTFX/events.json","paper":"https://pith.science/paper/KTPORQJT"},"agent_actions":{"view_html":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX","download_json":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX.json","view_paper":"https://pith.science/paper/KTPORQJT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.08191&json=true","fetch_graph":"https://pith.science/api/pith-number/KTPORQJTJEHHM7HQWPMUQHDTFX/graph.json","fetch_events":"https://pith.science/api/pith-number/KTPORQJTJEHHM7HQWPMUQHDTFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX/action/storage_attestation","attest_author":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX/action/author_attestation","sign_citation":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX/action/citation_signature","submit_replication":"https://pith.science/pith/KTPORQJTJEHHM7HQWPMUQHDTFX/action/replication_record"}},"created_at":"2026-07-05T02:39:42.782363+00:00","updated_at":"2026-07-05T02:39:42.782363+00:00"}