{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MXWGGZPXSFZW2NQKPGZXDUA6ID","short_pith_number":"pith:MXWGGZPX","schema_version":"1.0","canonical_sha256":"65ec6365f791736d360a79b371d01e40d7ebf1516db62bd50484b1be4a4da430","source":{"kind":"arxiv","id":"2302.05963","version":1},"attestation_state":"computed","paper":{"title":"Analyzing the Effectiveness of the Underlying Reasoning Tasks in Multi-hop Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akiko Aizawa, Anh-Khoa Duong Nguyen, Saku Sugawara, Xanh Ho","submitted_at":"2023-02-12T17:32:55Z","abstract_excerpt":"To explain the predicted answers and evaluate the reasoning abilities of models, several studies have utilized underlying reasoning (UR) tasks in multi-hop question answering (QA) datasets. However, it remains an open question as to how effective UR tasks are for the QA task when training models on both tasks in an end-to-end manner. In this study, we address this question by analyzing the effectiveness of UR tasks (including both sentence-level and entity-level tasks) in three aspects: (1) QA performance, (2) reasoning shortcuts, and (3) robustness. While the previous models have not been exp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.05963","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-12T17:32:55Z","cross_cats_sorted":[],"title_canon_sha256":"afd2656e4d4f2788f3a8d04088d7dc960f24309d6c03ebc0bd73db559e972e8d","abstract_canon_sha256":"79a614f658de95ddaa9dbeab4c69f0b3a3a3d55c2863045607ddb6402a20c8c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:41:00.677890Z","signature_b64":"Bn3VaXT79e2D+i1BQCHF6iFR6IjmqWg6zVA6J4ygDL1BuZjvxID7GPZOcWVD0E8pnIonCWkrxlWX73dbOTV/BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65ec6365f791736d360a79b371d01e40d7ebf1516db62bd50484b1be4a4da430","last_reissued_at":"2026-07-05T05:41:00.677487Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:41:00.677487Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing the Effectiveness of the Underlying Reasoning Tasks in Multi-hop Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akiko Aizawa, Anh-Khoa Duong Nguyen, Saku Sugawara, Xanh Ho","submitted_at":"2023-02-12T17:32:55Z","abstract_excerpt":"To explain the predicted answers and evaluate the reasoning abilities of models, several studies have utilized underlying reasoning (UR) tasks in multi-hop question answering (QA) datasets. However, it remains an open question as to how effective UR tasks are for the QA task when training models on both tasks in an end-to-end manner. In this study, we address this question by analyzing the effectiveness of UR tasks (including both sentence-level and entity-level tasks) in three aspects: (1) QA performance, (2) reasoning shortcuts, and (3) robustness. While the previous models have not been exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.05963","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.05963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.05963","created_at":"2026-07-05T05:41:00.677554+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.05963v1","created_at":"2026-07-05T05:41:00.677554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.05963","created_at":"2026-07-05T05:41:00.677554+00:00"},{"alias_kind":"pith_short_12","alias_value":"MXWGGZPXSFZW","created_at":"2026-07-05T05:41:00.677554+00:00"},{"alias_kind":"pith_short_16","alias_value":"MXWGGZPXSFZW2NQK","created_at":"2026-07-05T05:41:00.677554+00:00"},{"alias_kind":"pith_short_8","alias_value":"MXWGGZPX","created_at":"2026-07-05T05:41:00.677554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID","json":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID.json","graph_json":"https://pith.science/api/pith-number/MXWGGZPXSFZW2NQKPGZXDUA6ID/graph.json","events_json":"https://pith.science/api/pith-number/MXWGGZPXSFZW2NQKPGZXDUA6ID/events.json","paper":"https://pith.science/paper/MXWGGZPX"},"agent_actions":{"view_html":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID","download_json":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID.json","view_paper":"https://pith.science/paper/MXWGGZPX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.05963&json=true","fetch_graph":"https://pith.science/api/pith-number/MXWGGZPXSFZW2NQKPGZXDUA6ID/graph.json","fetch_events":"https://pith.science/api/pith-number/MXWGGZPXSFZW2NQKPGZXDUA6ID/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID/action/storage_attestation","attest_author":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID/action/author_attestation","sign_citation":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID/action/citation_signature","submit_replication":"https://pith.science/pith/MXWGGZPXSFZW2NQKPGZXDUA6ID/action/replication_record"}},"created_at":"2026-07-05T05:41:00.677554+00:00","updated_at":"2026-07-05T05:41:00.677554+00:00"}