{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:IZQ7P7RRZUOBCY4KWASH6FM4XK","short_pith_number":"pith:IZQ7P7RR","schema_version":"1.0","canonical_sha256":"4661f7fe31cd1c11638ab0247f159cbab6a78854813109aea626ddb95a36450a","source":{"kind":"arxiv","id":"2002.03932","version":1},"attestation_state":"computed","paper":{"title":"Pre-training Tasks for Embedding-based Large-scale Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.IR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Felix X. Yu, Sanjiv Kumar, Wei-Cheng Chang, Yiming Yang, Yin-Wen Chang","submitted_at":"2020-02-10T16:44:00Z","abstract_excerpt":"We consider the large-scale query-document retrieval problem: given a query (e.g., a question), return the set of relevant documents (e.g., paragraphs containing the answer) from a large document corpus. This problem is often solved in two steps. The retrieval phase first reduces the solution space, returning a subset of candidate documents. The scoring phase then re-ranks the documents. Critically, the retrieval algorithm not only desires high recall but also requires to be highly efficient, returning candidates in time sublinear to the number of documents. Unlike the scoring phase witnessing"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.03932","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-10T16:44:00Z","cross_cats_sorted":["cs.CL","cs.IR","stat.ML"],"title_canon_sha256":"5557308ba3aff06aef66e3a9b58dc45c4a0bf283f3d0c5be34512637dc86b920","abstract_canon_sha256":"68c6f729db32cbcb4972ede5f50cad5c65bbb06e44ac08497b01a79fc0e69549"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:39:27.671655Z","signature_b64":"sh6tp2MKiLGiSW6qEPZH/KoPd4uLcz82VBYr6OCjN+AKl1dYgL5txqVu0yYLhOzoOHpi4GwtNINWb47zDFkTCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4661f7fe31cd1c11638ab0247f159cbab6a78854813109aea626ddb95a36450a","last_reissued_at":"2026-07-05T00:39:27.671238Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:39:27.671238Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pre-training Tasks for Embedding-based Large-scale Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.IR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Felix X. Yu, Sanjiv Kumar, Wei-Cheng Chang, Yiming Yang, Yin-Wen Chang","submitted_at":"2020-02-10T16:44:00Z","abstract_excerpt":"We consider the large-scale query-document retrieval problem: given a query (e.g., a question), return the set of relevant documents (e.g., paragraphs containing the answer) from a large document corpus. This problem is often solved in two steps. The retrieval phase first reduces the solution space, returning a subset of candidate documents. The scoring phase then re-ranks the documents. Critically, the retrieval algorithm not only desires high recall but also requires to be highly efficient, returning candidates in time sublinear to the number of documents. Unlike the scoring phase witnessing"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.03932","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.03932/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.03932","created_at":"2026-07-05T00:39:27.671294+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.03932v1","created_at":"2026-07-05T00:39:27.671294+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.03932","created_at":"2026-07-05T00:39:27.671294+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZQ7P7RRZUOB","created_at":"2026-07-05T00:39:27.671294+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZQ7P7RRZUOBCY4K","created_at":"2026-07-05T00:39:27.671294+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZQ7P7RR","created_at":"2026-07-05T00:39:27.671294+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23962","citing_title":"From Index to Equity: Pre-Training Transformers for Stock Return Prediction","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2208.03299","citing_title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2112.09118","citing_title":"Unsupervised Dense Information Retrieval with Contrastive Learning","ref_index":119,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK","json":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK.json","graph_json":"https://pith.science/api/pith-number/IZQ7P7RRZUOBCY4KWASH6FM4XK/graph.json","events_json":"https://pith.science/api/pith-number/IZQ7P7RRZUOBCY4KWASH6FM4XK/events.json","paper":"https://pith.science/paper/IZQ7P7RR"},"agent_actions":{"view_html":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK","download_json":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK.json","view_paper":"https://pith.science/paper/IZQ7P7RR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.03932&json=true","fetch_graph":"https://pith.science/api/pith-number/IZQ7P7RRZUOBCY4KWASH6FM4XK/graph.json","fetch_events":"https://pith.science/api/pith-number/IZQ7P7RRZUOBCY4KWASH6FM4XK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK/action/storage_attestation","attest_author":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK/action/author_attestation","sign_citation":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK/action/citation_signature","submit_replication":"https://pith.science/pith/IZQ7P7RRZUOBCY4KWASH6FM4XK/action/replication_record"}},"created_at":"2026-07-05T00:39:27.671294+00:00","updated_at":"2026-07-05T00:39:27.671294+00:00"}