{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4E3QJ3Q3FGYJYYPQC5CT3DCGKE","short_pith_number":"pith:4E3QJ3Q3","schema_version":"1.0","canonical_sha256":"e13704ee1b29b09c61f017453d8c46510d812ea5875cb5933268535baa8e92ed","source":{"kind":"arxiv","id":"2406.06331","version":2},"attestation_state":"computed","paper":{"title":"MedExQA: Medical Question Answering Benchmark with Multiple Explanations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Honghan Wu, Jinge Wu, Yunsoo Kim, Yusuf Abdulle","submitted_at":"2024-06-10T14:47:04Z","abstract_excerpt":"This paper introduces MedExQA, a novel benchmark in medical question-answering, to evaluate large language models' (LLMs) understanding of medical knowledge through explanations. By constructing datasets across five distinct medical specialties that are underrepresented in current datasets and further incorporating multiple explanations for each question-answer pair, we address a major gap in current medical QA benchmarks which is the absence of comprehensive assessments of LLMs' ability to generate nuanced medical explanations. Our work highlights the importance of explainability in medical L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.06331","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-10T14:47:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4fb4ed0a71537786925f34290cf274f57dc61ead5120e6cc2b1c0f872613bd28","abstract_canon_sha256":"699f766724c2fe42d46f1b310a9db6374d65f54db844e39efe6eebb26ff05674"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:34.912264Z","signature_b64":"QTzreFY4+7VxsTqUXNcUt3NT+nOzS+6hUNxEcSPBAUeBlMjUYeIC5avvx7ivZ2k35YaUjObCpyhpmW6z1pS8Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e13704ee1b29b09c61f017453d8c46510d812ea5875cb5933268535baa8e92ed","last_reissued_at":"2026-07-05T08:39:34.911833Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:34.911833Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedExQA: Medical Question Answering Benchmark with Multiple Explanations","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Honghan Wu, Jinge Wu, Yunsoo Kim, Yusuf Abdulle","submitted_at":"2024-06-10T14:47:04Z","abstract_excerpt":"This paper introduces MedExQA, a novel benchmark in medical question-answering, to evaluate large language models' (LLMs) understanding of medical knowledge through explanations. By constructing datasets across five distinct medical specialties that are underrepresented in current datasets and further incorporating multiple explanations for each question-answer pair, we address a major gap in current medical QA benchmarks which is the absence of comprehensive assessments of LLMs' ability to generate nuanced medical explanations. Our work highlights the importance of explainability in medical L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.06331","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.06331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.06331","created_at":"2026-07-05T08:39:34.911887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.06331v2","created_at":"2026-07-05T08:39:34.911887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.06331","created_at":"2026-07-05T08:39:34.911887+00:00"},{"alias_kind":"pith_short_12","alias_value":"4E3QJ3Q3FGYJ","created_at":"2026-07-05T08:39:34.911887+00:00"},{"alias_kind":"pith_short_16","alias_value":"4E3QJ3Q3FGYJYYPQ","created_at":"2026-07-05T08:39:34.911887+00:00"},{"alias_kind":"pith_short_8","alias_value":"4E3QJ3Q3","created_at":"2026-07-05T08:39:34.911887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30637","citing_title":"EHRBench: An Automated and Reliable EHR-based Benchmark for Clinical Decision Making with LLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18111","citing_title":"How Good LLMs Are at Answering Bangla Medical Visual Questions? Dataset and Benchmarking","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20263","citing_title":"AROMA: Augmented Reasoning Over a Multimodal Architecture for Virtual Cell Genetic Perturbation Modeling","ref_index":94,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE","json":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE.json","graph_json":"https://pith.science/api/pith-number/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/graph.json","events_json":"https://pith.science/api/pith-number/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/events.json","paper":"https://pith.science/paper/4E3QJ3Q3"},"agent_actions":{"view_html":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE","download_json":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE.json","view_paper":"https://pith.science/paper/4E3QJ3Q3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.06331&json=true","fetch_graph":"https://pith.science/api/pith-number/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/graph.json","fetch_events":"https://pith.science/api/pith-number/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/action/storage_attestation","attest_author":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/action/author_attestation","sign_citation":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/action/citation_signature","submit_replication":"https://pith.science/pith/4E3QJ3Q3FGYJYYPQC5CT3DCGKE/action/replication_record"}},"created_at":"2026-07-05T08:39:34.911887+00:00","updated_at":"2026-07-05T08:39:34.911887+00:00"}