{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:C2U7ESLNXR7SVBQZDQXAFHIZ43","short_pith_number":"pith:C2U7ESLN","schema_version":"1.0","canonical_sha256":"16a9f2496dbc7f2a86191c2e029d19e6f7ffd3a3261b54f0013b53c17157f331","source":{"kind":"arxiv","id":"2109.03438","version":4},"attestation_state":"computed","paper":{"title":"ArchivalQA: A Large-scale Benchmark Dataset for Open Domain Question Answering over Historical News Collections","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adam Jatowt, Jiexin Wang, Masatoshi Yoshikawa","submitted_at":"2021-09-08T05:21:51Z","abstract_excerpt":"In the last few years, open-domain question answering (ODQA) has advanced rapidly due to the development of deep learning techniques and the availability of large-scale QA datasets. However, the current datasets are essentially designed for synchronic document collections (e.g., Wikipedia). Temporal news collections such as long-term news archives spanning several decades, are rarely used in training the models despite they are quite valuable for our society. To foster the research in the field of ODQA on such historical collections, we present ArchivalQA, a large question answering dataset co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.03438","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-09-08T05:21:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7c6d206819adbbaa2fe418a1ba6c066118e3f80d3c362c2d9e2e428e2974fba7","abstract_canon_sha256":"573e0ee9ac1b5b67e2af941c4e6ed54ed076e1b1dd4f502b4aa2239ca8424eb3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:58:53.477381Z","signature_b64":"BGGErEFAxPHIG1+7qsv87Mt0hecvI82hlnRzvhD+XAaa5rc6bdUoRTejoIpNhT2iMhtsh0I1CyMT8WwvPvO+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16a9f2496dbc7f2a86191c2e029d19e6f7ffd3a3261b54f0013b53c17157f331","last_reissued_at":"2026-07-05T03:58:53.476863Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:58:53.476863Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ArchivalQA: A Large-scale Benchmark Dataset for Open Domain Question Answering over Historical News Collections","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adam Jatowt, Jiexin Wang, Masatoshi Yoshikawa","submitted_at":"2021-09-08T05:21:51Z","abstract_excerpt":"In the last few years, open-domain question answering (ODQA) has advanced rapidly due to the development of deep learning techniques and the availability of large-scale QA datasets. However, the current datasets are essentially designed for synchronic document collections (e.g., Wikipedia). Temporal news collections such as long-term news archives spanning several decades, are rarely used in training the models despite they are quite valuable for our society. To foster the research in the field of ODQA on such historical collections, we present ArchivalQA, a large question answering dataset co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.03438","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.03438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.03438","created_at":"2026-07-05T03:58:53.476922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.03438v4","created_at":"2026-07-05T03:58:53.476922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.03438","created_at":"2026-07-05T03:58:53.476922+00:00"},{"alias_kind":"pith_short_12","alias_value":"C2U7ESLNXR7S","created_at":"2026-07-05T03:58:53.476922+00:00"},{"alias_kind":"pith_short_16","alias_value":"C2U7ESLNXR7SVBQZ","created_at":"2026-07-05T03:58:53.476922+00:00"},{"alias_kind":"pith_short_8","alias_value":"C2U7ESLN","created_at":"2026-07-05T03:58:53.476922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21783","citing_title":"Evaluating List Construction and Temporal Understanding capabilities of Large Language Models","ref_index":41,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43","json":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43.json","graph_json":"https://pith.science/api/pith-number/C2U7ESLNXR7SVBQZDQXAFHIZ43/graph.json","events_json":"https://pith.science/api/pith-number/C2U7ESLNXR7SVBQZDQXAFHIZ43/events.json","paper":"https://pith.science/paper/C2U7ESLN"},"agent_actions":{"view_html":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43","download_json":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43.json","view_paper":"https://pith.science/paper/C2U7ESLN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.03438&json=true","fetch_graph":"https://pith.science/api/pith-number/C2U7ESLNXR7SVBQZDQXAFHIZ43/graph.json","fetch_events":"https://pith.science/api/pith-number/C2U7ESLNXR7SVBQZDQXAFHIZ43/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43/action/storage_attestation","attest_author":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43/action/author_attestation","sign_citation":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43/action/citation_signature","submit_replication":"https://pith.science/pith/C2U7ESLNXR7SVBQZDQXAFHIZ43/action/replication_record"}},"created_at":"2026-07-05T03:58:53.476922+00:00","updated_at":"2026-07-05T03:58:53.476922+00:00"}