{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NLWTR4UJKCZ7QW57FALEBXN7OB","short_pith_number":"pith:NLWTR4UJ","schema_version":"1.0","canonical_sha256":"6aed38f28950b3f85bbf281640ddbf7042761b5a06b115645b3e7cd260a556e4","source":{"kind":"arxiv","id":"2404.05590","version":2},"attestation_state":"computed","paper":{"title":"MedExpQA: Multilingual Benchmarking of Large Language Models for Medical Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"I\\~nigo Alonso, Maite Oronoz, Rodrigo Agerri","submitted_at":"2024-04-08T15:03:57Z","abstract_excerpt":"Large Language Models (LLMs) have the potential of facilitating the development of Artificial Intelligence technology to assist medical experts for interactive decision support, which has been demonstrated by their competitive performances in Medical QA. However, while impressive, the required quality bar for medical applications remains far from being achieved. Currently, LLMs remain challenged by outdated knowledge and by their tendency to generate hallucinated content. Furthermore, most benchmarks to assess medical knowledge lack reference gold explanations which means that it is not possib"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05590","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-08T15:03:57Z","cross_cats_sorted":[],"title_canon_sha256":"62193eb0a9fea685ffc730d22d8a2bb985f3ff0b4af14f2dfee875e5270f6265","abstract_canon_sha256":"2cf8e3d5a4ab429cff48d30614466b530d5e9ebd9a37267521cf83d90d002e2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:30.613031Z","signature_b64":"rigXEdfSs8HKF6/01MwA71Pn1iX93KOl9lBjoe7C+n/8wqEYDsEf8K1LUaWVkFQY8vXeN/iOTlPNdzbYsIRRCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6aed38f28950b3f85bbf281640ddbf7042761b5a06b115645b3e7cd260a556e4","last_reissued_at":"2026-07-05T09:33:30.612550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:30.612550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedExpQA: Multilingual Benchmarking of Large Language Models for Medical Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"I\\~nigo Alonso, Maite Oronoz, Rodrigo Agerri","submitted_at":"2024-04-08T15:03:57Z","abstract_excerpt":"Large Language Models (LLMs) have the potential of facilitating the development of Artificial Intelligence technology to assist medical experts for interactive decision support, which has been demonstrated by their competitive performances in Medical QA. However, while impressive, the required quality bar for medical applications remains far from being achieved. Currently, LLMs remain challenged by outdated knowledge and by their tendency to generate hallucinated content. Furthermore, most benchmarks to assess medical knowledge lack reference gold explanations which means that it is not possib"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05590","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05590/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05590","created_at":"2026-07-05T09:33:30.612608+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05590v2","created_at":"2026-07-05T09:33:30.612608+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05590","created_at":"2026-07-05T09:33:30.612608+00:00"},{"alias_kind":"pith_short_12","alias_value":"NLWTR4UJKCZ7","created_at":"2026-07-05T09:33:30.612608+00:00"},{"alias_kind":"pith_short_16","alias_value":"NLWTR4UJKCZ7QW57","created_at":"2026-07-05T09:33:30.612608+00:00"},{"alias_kind":"pith_short_8","alias_value":"NLWTR4UJ","created_at":"2026-07-05T09:33:30.612608+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03305","citing_title":"The Reliability Gap in Benchmark Auditing: Distribution Shift and Scale as Failure Modes of Contamination Detection","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03305","citing_title":"The Reliability Gap in Benchmark Auditing: Distribution Shift and Scale as Failure Modes of Contamination Detection","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB","json":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB.json","graph_json":"https://pith.science/api/pith-number/NLWTR4UJKCZ7QW57FALEBXN7OB/graph.json","events_json":"https://pith.science/api/pith-number/NLWTR4UJKCZ7QW57FALEBXN7OB/events.json","paper":"https://pith.science/paper/NLWTR4UJ"},"agent_actions":{"view_html":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB","download_json":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB.json","view_paper":"https://pith.science/paper/NLWTR4UJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05590&json=true","fetch_graph":"https://pith.science/api/pith-number/NLWTR4UJKCZ7QW57FALEBXN7OB/graph.json","fetch_events":"https://pith.science/api/pith-number/NLWTR4UJKCZ7QW57FALEBXN7OB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB/action/storage_attestation","attest_author":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB/action/author_attestation","sign_citation":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB/action/citation_signature","submit_replication":"https://pith.science/pith/NLWTR4UJKCZ7QW57FALEBXN7OB/action/replication_record"}},"created_at":"2026-07-05T09:33:30.612608+00:00","updated_at":"2026-07-05T09:33:30.612608+00:00"}