{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7Q4Z7YPD3UGZRCL6ZYRFHS3VN5","short_pith_number":"pith:7Q4Z7YPD","schema_version":"1.0","canonical_sha256":"fc399fe1e3dd0d98897ece2253cb756f582da29139f8c634e08e551ccb417426","source":{"kind":"arxiv","id":"2409.01497","version":2},"attestation_state":"computed","paper":{"title":"DiversityMedQA: Assessing Demographic Biases in Medical Diagnosis using Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dhiyaan Nirmal, Dhruv Alamuri, Hudson McBride, Jong Moon, Kevin Zhu, Rajarshi Ghosh, Rajat Rawat, Sean O'Brien","submitted_at":"2024-09-02T23:37:20Z","abstract_excerpt":"As large language models (LLMs) gain traction in healthcare, concerns about their susceptibility to demographic biases are growing. We introduce {DiversityMedQA}, a novel benchmark designed to assess LLM responses to medical queries across diverse patient demographics, such as gender and ethnicity. By perturbing questions from the MedQA dataset, which comprises medical board exam questions, we created a benchmark that captures the nuanced differences in medical diagnosis across varying patient profiles. Our findings reveal notable discrepancies in model performance when tested against these de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.01497","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-02T23:37:20Z","cross_cats_sorted":[],"title_canon_sha256":"5b40c6e34ffba4fbcfda77e994253ec65b5b3df1c42abcfa77a80cd19ffff9a0","abstract_canon_sha256":"1a642684719931a1fe5a87d9db5e67c20c234769b35ac0a16a26aff5181724bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:14.117764Z","signature_b64":"vGikqRhymVN0ydkMHGTwuLsgMHCC/9Rd2+AaInIDq3PpIMfbmnmE/9zFmTRO5GWYWfss8RStFMq2UdPnu1RcAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc399fe1e3dd0d98897ece2253cb756f582da29139f8c634e08e551ccb417426","last_reissued_at":"2026-07-05T09:45:14.117294Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:14.117294Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiversityMedQA: Assessing Demographic Biases in Medical Diagnosis using Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dhiyaan Nirmal, Dhruv Alamuri, Hudson McBride, Jong Moon, Kevin Zhu, Rajarshi Ghosh, Rajat Rawat, Sean O'Brien","submitted_at":"2024-09-02T23:37:20Z","abstract_excerpt":"As large language models (LLMs) gain traction in healthcare, concerns about their susceptibility to demographic biases are growing. We introduce {DiversityMedQA}, a novel benchmark designed to assess LLM responses to medical queries across diverse patient demographics, such as gender and ethnicity. By perturbing questions from the MedQA dataset, which comprises medical board exam questions, we created a benchmark that captures the nuanced differences in medical diagnosis across varying patient profiles. Our findings reveal notable discrepancies in model performance when tested against these de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.01497","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.01497/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.01497","created_at":"2026-07-05T09:45:14.117353+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.01497v2","created_at":"2026-07-05T09:45:14.117353+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.01497","created_at":"2026-07-05T09:45:14.117353+00:00"},{"alias_kind":"pith_short_12","alias_value":"7Q4Z7YPD3UGZ","created_at":"2026-07-05T09:45:14.117353+00:00"},{"alias_kind":"pith_short_16","alias_value":"7Q4Z7YPD3UGZRCL6","created_at":"2026-07-05T09:45:14.117353+00:00"},{"alias_kind":"pith_short_8","alias_value":"7Q4Z7YPD","created_at":"2026-07-05T09:45:14.117353+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01048","citing_title":"Compared to What? Baselines and Metrics for Counterfactual Prompting","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5","json":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5.json","graph_json":"https://pith.science/api/pith-number/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/graph.json","events_json":"https://pith.science/api/pith-number/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/events.json","paper":"https://pith.science/paper/7Q4Z7YPD"},"agent_actions":{"view_html":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5","download_json":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5.json","view_paper":"https://pith.science/paper/7Q4Z7YPD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.01497&json=true","fetch_graph":"https://pith.science/api/pith-number/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/graph.json","fetch_events":"https://pith.science/api/pith-number/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/action/storage_attestation","attest_author":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/action/author_attestation","sign_citation":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/action/citation_signature","submit_replication":"https://pith.science/pith/7Q4Z7YPD3UGZRCL6ZYRFHS3VN5/action/replication_record"}},"created_at":"2026-07-05T09:45:14.117353+00:00","updated_at":"2026-07-05T09:45:14.117353+00:00"}