{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:AXCTVTASUWAI2NRKXSSEE64EXS","short_pith_number":"pith:AXCTVTAS","schema_version":"1.0","canonical_sha256":"05c53acc12a5808d362abca4427b84bcbccc61727216fc4f7be92a22e79ce923","source":{"kind":"arxiv","id":"2004.05484","version":1},"attestation_state":"computed","paper":{"title":"LAReQA: Language-agnostic answer retrieval from a multilingual pool","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Phillips, Aditya Barua, Noah Constant, Rami Al-Rfou, Uma Roy, Yinfei Yang","submitted_at":"2020-04-11T20:51:11Z","abstract_excerpt":"We present LAReQA, a challenging new benchmark for language-agnostic answer retrieval from a multilingual candidate pool. Unlike previous cross-lingual tasks, LAReQA tests for \"strong\" cross-lingual alignment, requiring semantically related cross-language pairs to be closer in representation space than unrelated same-language pairs. Building on multilingual BERT (mBERT), we study different strategies for achieving strong alignment. We find that augmenting training data via machine translation is effective, and improves significantly over using mBERT out-of-the-box. Interestingly, the embedding"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.05484","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-11T20:51:11Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e6866bfaa64845a7221dc0296d43bf9f747acaa457a0fb42a624b036e96b9787","abstract_canon_sha256":"5b9671687dd92317259617d250d91db1404950ab798213c42e499b6787cf9bdc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:54:37.542503Z","signature_b64":"moUd3p7xfRIZILDXVm78Dd/1AmnIB1LVtzZXwEnbjd+/PIl7txf2KQoI9gzSynZxwevBy/d60nsdCNgY/nsLDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"05c53acc12a5808d362abca4427b84bcbccc61727216fc4f7be92a22e79ce923","last_reissued_at":"2026-07-05T00:54:37.542060Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:54:37.542060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LAReQA: Language-agnostic answer retrieval from a multilingual pool","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Phillips, Aditya Barua, Noah Constant, Rami Al-Rfou, Uma Roy, Yinfei Yang","submitted_at":"2020-04-11T20:51:11Z","abstract_excerpt":"We present LAReQA, a challenging new benchmark for language-agnostic answer retrieval from a multilingual candidate pool. Unlike previous cross-lingual tasks, LAReQA tests for \"strong\" cross-lingual alignment, requiring semantically related cross-language pairs to be closer in representation space than unrelated same-language pairs. Building on multilingual BERT (mBERT), we study different strategies for achieving strong alignment. We find that augmenting training data via machine translation is effective, and improves significantly over using mBERT out-of-the-box. Interestingly, the embedding"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.05484","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.05484/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.05484","created_at":"2026-07-05T00:54:37.542125+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.05484v1","created_at":"2026-07-05T00:54:37.542125+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.05484","created_at":"2026-07-05T00:54:37.542125+00:00"},{"alias_kind":"pith_short_12","alias_value":"AXCTVTASUWAI","created_at":"2026-07-05T00:54:37.542125+00:00"},{"alias_kind":"pith_short_16","alias_value":"AXCTVTASUWAI2NRK","created_at":"2026-07-05T00:54:37.542125+00:00"},{"alias_kind":"pith_short_8","alias_value":"AXCTVTAS","created_at":"2026-07-05T00:54:37.542125+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17709","citing_title":"TyDi QA-WANA: A Benchmark for Information-Seeking Question Answering in Languages of West Asia and North Africa","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS","json":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS.json","graph_json":"https://pith.science/api/pith-number/AXCTVTASUWAI2NRKXSSEE64EXS/graph.json","events_json":"https://pith.science/api/pith-number/AXCTVTASUWAI2NRKXSSEE64EXS/events.json","paper":"https://pith.science/paper/AXCTVTAS"},"agent_actions":{"view_html":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS","download_json":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS.json","view_paper":"https://pith.science/paper/AXCTVTAS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.05484&json=true","fetch_graph":"https://pith.science/api/pith-number/AXCTVTASUWAI2NRKXSSEE64EXS/graph.json","fetch_events":"https://pith.science/api/pith-number/AXCTVTASUWAI2NRKXSSEE64EXS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS/action/storage_attestation","attest_author":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS/action/author_attestation","sign_citation":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS/action/citation_signature","submit_replication":"https://pith.science/pith/AXCTVTASUWAI2NRKXSSEE64EXS/action/replication_record"}},"created_at":"2026-07-05T00:54:37.542125+00:00","updated_at":"2026-07-05T00:54:37.542125+00:00"}