{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RHIQHW36MG7TLSHKEIN4LMN252","short_pith_number":"pith:RHIQHW36","schema_version":"1.0","canonical_sha256":"89d103db7e61bf35c8ea221bc5b1baee8795e54009a65094948c96d5576db7ce","source":{"kind":"arxiv","id":"2406.09923","version":2},"attestation_state":"computed","paper":{"title":"CliBench: A Multifaceted and Multigranular Evaluation of Large Language Models for Clinical Decision Making","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenchen Ye, Mingyu Derek Ma, Peipei Ping, Timothy S Chang, Wei Wang, Xiaoxuan Wang, Yu Yan","submitted_at":"2024-06-14T11:10:17Z","abstract_excerpt":"The integration of Artificial Intelligence (AI), especially Large Language Models (LLMs), into the clinical diagnosis process offers significant potential to improve the efficiency and accessibility of medical care. While LLMs have shown some promise in the medical domain, their application in clinical diagnosis remains underexplored, especially in real-world clinical practice, where highly sophisticated, patient-specific decisions need to be made. Current evaluations of LLMs in this field are often narrow in scope, focusing on specific diseases or specialties and employing simplified diagnost"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09923","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-14T11:10:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"cfc1452da2bc3a428a14b0e5961e2242b747c355bafd48f2d26f272e62c8a640","abstract_canon_sha256":"62813088403e46f1db2cd409cca3fded176032fee34a04276474d8a9a79d9a20"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:29.106119Z","signature_b64":"GLkTkIEH7qtq7H+RKrsG4OMQ1XfRJd0wkdT8n/DP6I9Tr5FwRqBaWVymo5pfwSCq7sd2OmvFjCSrTIE7BZXFAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89d103db7e61bf35c8ea221bc5b1baee8795e54009a65094948c96d5576db7ce","last_reissued_at":"2026-07-05T09:19:29.105634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:29.105634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CliBench: A Multifaceted and Multigranular Evaluation of Large Language Models for Clinical Decision Making","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenchen Ye, Mingyu Derek Ma, Peipei Ping, Timothy S Chang, Wei Wang, Xiaoxuan Wang, Yu Yan","submitted_at":"2024-06-14T11:10:17Z","abstract_excerpt":"The integration of Artificial Intelligence (AI), especially Large Language Models (LLMs), into the clinical diagnosis process offers significant potential to improve the efficiency and accessibility of medical care. While LLMs have shown some promise in the medical domain, their application in clinical diagnosis remains underexplored, especially in real-world clinical practice, where highly sophisticated, patient-specific decisions need to be made. Current evaluations of LLMs in this field are often narrow in scope, focusing on specific diseases or specialties and employing simplified diagnost"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09923","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09923","created_at":"2026-07-05T09:19:29.105695+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09923v2","created_at":"2026-07-05T09:19:29.105695+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09923","created_at":"2026-07-05T09:19:29.105695+00:00"},{"alias_kind":"pith_short_12","alias_value":"RHIQHW36MG7T","created_at":"2026-07-05T09:19:29.105695+00:00"},{"alias_kind":"pith_short_16","alias_value":"RHIQHW36MG7TLSHK","created_at":"2026-07-05T09:19:29.105695+00:00"},{"alias_kind":"pith_short_8","alias_value":"RHIQHW36","created_at":"2026-07-05T09:19:29.105695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03543","citing_title":"D2MDT: Department-aware Multidisciplinary Team Consultation with Deliberation for Efficient Clinical Prediction","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01301","citing_title":"Med-HEAL: Analyzing and Mitigating Hallucinations in Medical LLMs with Hallucination-Aware In-Context Learning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14543","citing_title":"RxEval: A Prescription-Level Benchmark for Evaluating LLM Medication Recommendation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13542","citing_title":"RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252","json":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252.json","graph_json":"https://pith.science/api/pith-number/RHIQHW36MG7TLSHKEIN4LMN252/graph.json","events_json":"https://pith.science/api/pith-number/RHIQHW36MG7TLSHKEIN4LMN252/events.json","paper":"https://pith.science/paper/RHIQHW36"},"agent_actions":{"view_html":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252","download_json":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252.json","view_paper":"https://pith.science/paper/RHIQHW36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09923&json=true","fetch_graph":"https://pith.science/api/pith-number/RHIQHW36MG7TLSHKEIN4LMN252/graph.json","fetch_events":"https://pith.science/api/pith-number/RHIQHW36MG7TLSHKEIN4LMN252/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252/action/storage_attestation","attest_author":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252/action/author_attestation","sign_citation":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252/action/citation_signature","submit_replication":"https://pith.science/pith/RHIQHW36MG7TLSHKEIN4LMN252/action/replication_record"}},"created_at":"2026-07-05T09:19:29.105695+00:00","updated_at":"2026-07-05T09:19:29.105695+00:00"}