{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XG5XCWGUELO75DRZXGSGDI54ZH","short_pith_number":"pith:XG5XCWGU","schema_version":"1.0","canonical_sha256":"b9bb7158d422ddfe8e39b9a461a3bcc9ca1d57a316e28abc5ad53cc3fccf3ea3","source":{"kind":"arxiv","id":"2503.15454","version":3},"attestation_state":"computed","paper":{"title":"Bias Evaluation and Mitigation in Retrieval-Augmented Medical Question-Answering Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Zhang, Yanshan Wang, Yuelyu Ji","submitted_at":"2025-03-19T17:36:35Z","abstract_excerpt":"Medical Question Answering systems based on Retrieval Augmented Generation is promising for clinical decision support because they can integrate external knowledge, thus reducing inaccuracies inherent in standalone large language models (LLMs). However, these systems may unintentionally propagate or amplify biases associated with sensitive demographic attributes like race, gender, and socioeconomic factors. This study systematically evaluates demographic biases within medical RAG pipelines across multiple QA benchmarks, including MedQA, MedMCQA, MMLU, and EquityMedQA. We quantify disparities i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.15454","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-19T17:36:35Z","cross_cats_sorted":[],"title_canon_sha256":"cc73192e705eac23a4e82bdb513609155ba8460dddee6871380211a9385a93a1","abstract_canon_sha256":"0d9556971dfc60bcf9dad831447cc3b55cc759354dd7cf340bc0b46f2022fe73"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:47.798813Z","signature_b64":"Bf42JZ+oLUSnC2A7JcY9iS38hb9Ofge1WzuMfx2rAthGfVuWjwC+bvW4LAMLZ2+iEzzvZUxgGPWpY1jg3E9UAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9bb7158d422ddfe8e39b9a461a3bcc9ca1d57a316e28abc5ad53cc3fccf3ea3","last_reissued_at":"2026-07-05T10:39:47.798284Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:47.798284Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bias Evaluation and Mitigation in Retrieval-Augmented Medical Question-Answering Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Zhang, Yanshan Wang, Yuelyu Ji","submitted_at":"2025-03-19T17:36:35Z","abstract_excerpt":"Medical Question Answering systems based on Retrieval Augmented Generation is promising for clinical decision support because they can integrate external knowledge, thus reducing inaccuracies inherent in standalone large language models (LLMs). However, these systems may unintentionally propagate or amplify biases associated with sensitive demographic attributes like race, gender, and socioeconomic factors. This study systematically evaluates demographic biases within medical RAG pipelines across multiple QA benchmarks, including MedQA, MedMCQA, MMLU, and EquityMedQA. We quantify disparities i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.15454","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.15454/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.15454","created_at":"2026-07-05T10:39:47.798362+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.15454v3","created_at":"2026-07-05T10:39:47.798362+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.15454","created_at":"2026-07-05T10:39:47.798362+00:00"},{"alias_kind":"pith_short_12","alias_value":"XG5XCWGUELO7","created_at":"2026-07-05T10:39:47.798362+00:00"},{"alias_kind":"pith_short_16","alias_value":"XG5XCWGUELO75DRZ","created_at":"2026-07-05T10:39:47.798362+00:00"},{"alias_kind":"pith_short_8","alias_value":"XG5XCWGU","created_at":"2026-07-05T10:39:47.798362+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":265,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18806","citing_title":"Towards FairRAG: Preventing Representational Harm in Retrieval-Augmented Generation by Enforcing Fair Exposure at Retrieval Time","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH","json":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH.json","graph_json":"https://pith.science/api/pith-number/XG5XCWGUELO75DRZXGSGDI54ZH/graph.json","events_json":"https://pith.science/api/pith-number/XG5XCWGUELO75DRZXGSGDI54ZH/events.json","paper":"https://pith.science/paper/XG5XCWGU"},"agent_actions":{"view_html":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH","download_json":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH.json","view_paper":"https://pith.science/paper/XG5XCWGU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.15454&json=true","fetch_graph":"https://pith.science/api/pith-number/XG5XCWGUELO75DRZXGSGDI54ZH/graph.json","fetch_events":"https://pith.science/api/pith-number/XG5XCWGUELO75DRZXGSGDI54ZH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH/action/storage_attestation","attest_author":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH/action/author_attestation","sign_citation":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH/action/citation_signature","submit_replication":"https://pith.science/pith/XG5XCWGUELO75DRZXGSGDI54ZH/action/replication_record"}},"created_at":"2026-07-05T10:39:47.798362+00:00","updated_at":"2026-07-05T10:39:47.798362+00:00"}