{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:UNMASM224ZVJNNT7D6XRK7ZQEL","short_pith_number":"pith:UNMASM22","schema_version":"1.0","canonical_sha256":"a35809335ae66a96b67f1faf157f3022fc90f9dcb2298f39dd6859276e9a9c1a","source":{"kind":"arxiv","id":"2008.02637","version":1},"attestation_state":"computed","paper":{"title":"Question and Answer Test-Train Overlap in Open-Domain Question Answering Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Patrick Lewis, Pontus Stenetorp, Sebastian Riedel","submitted_at":"2020-08-06T13:17:43Z","abstract_excerpt":"Ideally Open-Domain Question Answering models should exhibit a number of competencies, ranging from simply memorizing questions seen at training time, to answering novel question formulations with answers seen during training, to generalizing to completely novel questions with novel answers. However, single aggregated test set scores do not show the full picture of what capabilities models truly have. In this work, we perform a detailed study of the test sets of three popular open-domain benchmark datasets with respect to these competencies. We find that 60-70% of test-time answers are also pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.02637","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-08-06T13:17:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"58b4fc662ae73582f4ed5254fababd62a4f5f7c38194b7ea1ac7d32ecca27548","abstract_canon_sha256":"03b36503b34d5e034ec0cba3b1ebe4758cd21dbdb9d8481dbdc1dfaa04819a6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:25:15.866354Z","signature_b64":"5UtMnR/G6HI4Z3ICKWqHit0GDoh+03ECLs59PzXM9BoDSLajtLhAygnBjkGC7Yhy1d0G7fd6msnbRX5ERUw1AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a35809335ae66a96b67f1faf157f3022fc90f9dcb2298f39dd6859276e9a9c1a","last_reissued_at":"2026-07-05T01:25:15.865982Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:25:15.865982Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Question and Answer Test-Train Overlap in Open-Domain Question Answering Datasets","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Patrick Lewis, Pontus Stenetorp, Sebastian Riedel","submitted_at":"2020-08-06T13:17:43Z","abstract_excerpt":"Ideally Open-Domain Question Answering models should exhibit a number of competencies, ranging from simply memorizing questions seen at training time, to answering novel question formulations with answers seen during training, to generalizing to completely novel questions with novel answers. However, single aggregated test set scores do not show the full picture of what capabilities models truly have. In this work, we perform a detailed study of the test sets of three popular open-domain benchmark datasets with respect to these competencies. We find that 60-70% of test-time answers are also pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.02637","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.02637/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.02637","created_at":"2026-07-05T01:25:15.866030+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.02637v1","created_at":"2026-07-05T01:25:15.866030+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.02637","created_at":"2026-07-05T01:25:15.866030+00:00"},{"alias_kind":"pith_short_12","alias_value":"UNMASM224ZVJ","created_at":"2026-07-05T01:25:15.866030+00:00"},{"alias_kind":"pith_short_16","alias_value":"UNMASM224ZVJNNT7","created_at":"2026-07-05T01:25:15.866030+00:00"},{"alias_kind":"pith_short_8","alias_value":"UNMASM22","created_at":"2026-07-05T01:25:15.866030+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.03387","citing_title":"LIMO: Less is More for Reasoning","ref_index":251,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11511","citing_title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2112.04359","citing_title":"Ethical and social risks of harm from Language Models","ref_index":162,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL","json":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL.json","graph_json":"https://pith.science/api/pith-number/UNMASM224ZVJNNT7D6XRK7ZQEL/graph.json","events_json":"https://pith.science/api/pith-number/UNMASM224ZVJNNT7D6XRK7ZQEL/events.json","paper":"https://pith.science/paper/UNMASM22"},"agent_actions":{"view_html":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL","download_json":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL.json","view_paper":"https://pith.science/paper/UNMASM22","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.02637&json=true","fetch_graph":"https://pith.science/api/pith-number/UNMASM224ZVJNNT7D6XRK7ZQEL/graph.json","fetch_events":"https://pith.science/api/pith-number/UNMASM224ZVJNNT7D6XRK7ZQEL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL/action/storage_attestation","attest_author":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL/action/author_attestation","sign_citation":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL/action/citation_signature","submit_replication":"https://pith.science/pith/UNMASM224ZVJNNT7D6XRK7ZQEL/action/replication_record"}},"created_at":"2026-07-05T01:25:15.866030+00:00","updated_at":"2026-07-05T01:25:15.866030+00:00"}