{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M5MHBJGFNZBOHPHHWNKC4C5ZBU","short_pith_number":"pith:M5MHBJGF","schema_version":"1.0","canonical_sha256":"675870a4c56e42e3bce7b3542e0bb90d1220994650fd8c881ef32fce58b7bc61","source":{"kind":"arxiv","id":"2407.05250","version":2},"attestation_state":"computed","paper":{"title":"CLIMB: A Benchmark of Clinical Bias in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jieyu Zhao, Mingyu Derek Ma, Muhao Chen, Shudi Hou, Wei Wang, Yubo Zhang","submitted_at":"2024-07-07T03:41:51Z","abstract_excerpt":"Large language models (LLMs) are increasingly applied to clinical decision-making. However, their potential to exhibit bias poses significant risks to clinical equity. Currently, there is a lack of benchmarks that systematically evaluate such clinical bias in LLMs. While in downstream tasks, some biases of LLMs can be avoided such as by instructing the model to answer \"I'm not sure...\", the internal bias hidden within the model still lacks deep studies. We introduce CLIMB (shorthand for A Benchmark of Clinical Bias in Large Language Models), a pioneering comprehensive benchmark to evaluate bot"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05250","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-07T03:41:51Z","cross_cats_sorted":[],"title_canon_sha256":"b365c11db7226851deba3dcb3abcdf6755ca09bdaf828b7af13535525acbcf8d","abstract_canon_sha256":"36a0763862e14d7f71b3a075b226c1333842c85cb3f749ada81adc3556c7355d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:34.650635Z","signature_b64":"fyx8MU6Q9hZArbBZz0jDx2dnUzzfkobnIWlswY5/kYDQXmDdU2U3cLpVZVdJDoJd/voSMz9boYYofGfnkIFuBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"675870a4c56e42e3bce7b3542e0bb90d1220994650fd8c881ef32fce58b7bc61","last_reissued_at":"2026-07-05T09:35:34.650137Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:34.650137Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIMB: A Benchmark of Clinical Bias in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jieyu Zhao, Mingyu Derek Ma, Muhao Chen, Shudi Hou, Wei Wang, Yubo Zhang","submitted_at":"2024-07-07T03:41:51Z","abstract_excerpt":"Large language models (LLMs) are increasingly applied to clinical decision-making. However, their potential to exhibit bias poses significant risks to clinical equity. Currently, there is a lack of benchmarks that systematically evaluate such clinical bias in LLMs. While in downstream tasks, some biases of LLMs can be avoided such as by instructing the model to answer \"I'm not sure...\", the internal bias hidden within the model still lacks deep studies. We introduce CLIMB (shorthand for A Benchmark of Clinical Bias in Large Language Models), a pioneering comprehensive benchmark to evaluate bot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05250","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05250/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05250","created_at":"2026-07-05T09:35:34.650197+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05250v2","created_at":"2026-07-05T09:35:34.650197+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05250","created_at":"2026-07-05T09:35:34.650197+00:00"},{"alias_kind":"pith_short_12","alias_value":"M5MHBJGFNZBO","created_at":"2026-07-05T09:35:34.650197+00:00"},{"alias_kind":"pith_short_16","alias_value":"M5MHBJGFNZBOHPHH","created_at":"2026-07-05T09:35:34.650197+00:00"},{"alias_kind":"pith_short_8","alias_value":"M5MHBJGF","created_at":"2026-07-05T09:35:34.650197+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10852","citing_title":"LLMs on Trial: Evaluating Judicial Fairness for Large Language Models","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU","json":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU.json","graph_json":"https://pith.science/api/pith-number/M5MHBJGFNZBOHPHHWNKC4C5ZBU/graph.json","events_json":"https://pith.science/api/pith-number/M5MHBJGFNZBOHPHHWNKC4C5ZBU/events.json","paper":"https://pith.science/paper/M5MHBJGF"},"agent_actions":{"view_html":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU","download_json":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU.json","view_paper":"https://pith.science/paper/M5MHBJGF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05250&json=true","fetch_graph":"https://pith.science/api/pith-number/M5MHBJGFNZBOHPHHWNKC4C5ZBU/graph.json","fetch_events":"https://pith.science/api/pith-number/M5MHBJGFNZBOHPHHWNKC4C5ZBU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU/action/storage_attestation","attest_author":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU/action/author_attestation","sign_citation":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU/action/citation_signature","submit_replication":"https://pith.science/pith/M5MHBJGFNZBOHPHHWNKC4C5ZBU/action/replication_record"}},"created_at":"2026-07-05T09:35:34.650197+00:00","updated_at":"2026-07-05T09:35:34.650197+00:00"}