{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZJE3D3FKDO4WN64IRKWRGSFWE7","short_pith_number":"pith:ZJE3D3FK","schema_version":"1.0","canonical_sha256":"ca49b1ecaa1bb966fb888aad1348b627e1fbb1bdf48cf56524ef2be927911fae","source":{"kind":"arxiv","id":"2406.13397","version":1},"attestation_state":"computed","paper":{"title":"MoreHopQA: More Than Multi-hop Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akiko Aizawa, Florian Boudin, Jiahao Huang, Julian Schnitzler, Saku Sugawara, Xanh Ho","submitted_at":"2024-06-19T09:38:59Z","abstract_excerpt":"Most existing multi-hop datasets are extractive answer datasets, where the answers to the questions can be extracted directly from the provided context. This often leads models to use heuristics or shortcuts instead of performing true multi-hop reasoning. In this paper, we propose a new multi-hop dataset, MoreHopQA, which shifts from extractive to generative answers. Our dataset is created by utilizing three existing multi-hop datasets: HotpotQA, 2WikiMultihopQA, and MuSiQue. Instead of relying solely on factual reasoning, we enhance the existing multi-hop questions by adding another layer of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.13397","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-19T09:38:59Z","cross_cats_sorted":[],"title_canon_sha256":"e8e900368505c2f6c37323a6a6f046a168429940088b3c309d9e638d7ea7c13e","abstract_canon_sha256":"734cf9c7bfebb76e0ceec9a4d3594a6cd48274ec39cf7e924bc6e2301281e322"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:29.585866Z","signature_b64":"7FgcyAfueMTvuD3l/rNpYGYCPGRGRDaKZdJh5EjPl/+yiNN9m9Xs2eeZMvePPr0jfY63aN3saKVgdQjftPvSDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca49b1ecaa1bb966fb888aad1348b627e1fbb1bdf48cf56524ef2be927911fae","last_reissued_at":"2026-07-05T08:34:29.585379Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:29.585379Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoreHopQA: More Than Multi-hop Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akiko Aizawa, Florian Boudin, Jiahao Huang, Julian Schnitzler, Saku Sugawara, Xanh Ho","submitted_at":"2024-06-19T09:38:59Z","abstract_excerpt":"Most existing multi-hop datasets are extractive answer datasets, where the answers to the questions can be extracted directly from the provided context. This often leads models to use heuristics or shortcuts instead of performing true multi-hop reasoning. In this paper, we propose a new multi-hop dataset, MoreHopQA, which shifts from extractive to generative answers. Our dataset is created by utilizing three existing multi-hop datasets: HotpotQA, 2WikiMultihopQA, and MuSiQue. Instead of relying solely on factual reasoning, we enhance the existing multi-hop questions by adding another layer of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.13397","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.13397/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.13397","created_at":"2026-07-05T08:34:29.585438+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.13397v1","created_at":"2026-07-05T08:34:29.585438+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.13397","created_at":"2026-07-05T08:34:29.585438+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZJE3D3FKDO4W","created_at":"2026-07-05T08:34:29.585438+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZJE3D3FKDO4WN64I","created_at":"2026-07-05T08:34:29.585438+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZJE3D3FK","created_at":"2026-07-05T08:34:29.585438+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05901","citing_title":"Reducing Hallucinations in Complex Question Answering using Simple Graph-based Retrieval-Augmented Generation (long version)","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22358","citing_title":"Integrating Chain-of-Thought into Generative Retrieval: A Preliminary Study","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19228","citing_title":"Diagnosing Multi-step Reasoning Failures in Black-box LLMs via Stepwise Confidence Attribution","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09721","citing_title":"Jamendo-MT-QA: A Benchmark for Multi-Track Comparative Music Question Answering","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7","json":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7.json","graph_json":"https://pith.science/api/pith-number/ZJE3D3FKDO4WN64IRKWRGSFWE7/graph.json","events_json":"https://pith.science/api/pith-number/ZJE3D3FKDO4WN64IRKWRGSFWE7/events.json","paper":"https://pith.science/paper/ZJE3D3FK"},"agent_actions":{"view_html":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7","download_json":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7.json","view_paper":"https://pith.science/paper/ZJE3D3FK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.13397&json=true","fetch_graph":"https://pith.science/api/pith-number/ZJE3D3FKDO4WN64IRKWRGSFWE7/graph.json","fetch_events":"https://pith.science/api/pith-number/ZJE3D3FKDO4WN64IRKWRGSFWE7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7/action/storage_attestation","attest_author":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7/action/author_attestation","sign_citation":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7/action/citation_signature","submit_replication":"https://pith.science/pith/ZJE3D3FKDO4WN64IRKWRGSFWE7/action/replication_record"}},"created_at":"2026-07-05T08:34:29.585438+00:00","updated_at":"2026-07-05T08:34:29.585438+00:00"}