{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SPPDH3COJVSXSKL6CGMZ4CDTT2","short_pith_number":"pith:SPPDH3CO","schema_version":"1.0","canonical_sha256":"93de33ec4e4d6579297e11999e08739ea195d37da1d5ae362d6f158707ab7648","source":{"kind":"arxiv","id":"2212.13138","version":1},"attestation_state":"computed","paper":{"title":"Large Language Models Encode Clinical Knowledge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aakanksha Chowdhery, Ajay Tanwani, Alan Karthikesalingam, Alvin Rajkomar, Blaise Aguera y Arcas, Chris Kelly, Christopher Semturs, Dale Webster, Greg S. Corrado, Heather Cole-Lewis, Hyung Won Chung, Jason Wei, Joelle Barral, Juraj Gottweis, Karan Singhal, Katherine Chou, Martin Seneviratne, Nathaneal Scharli, Nathan Scales, Nenad Tomasev, Paul Gamble, Perry Payne, Philip Mansfield, Shekoofeh Azizi, S. Sara Mahdavi, Stephen Pfohl, Tao Tu, Vivek Natarajan, Yossi Matias, Yun Liu","submitted_at":"2022-12-26T14:28:24Z","abstract_excerpt":"Large language models (LLMs) have demonstrated impressive capabilities in natural language understanding and generation, but the quality bar for medical and clinical applications is high. Today, attempts to assess models' clinical knowledge typically rely on automated evaluations on limited benchmarks. There is no standard to evaluate model predictions and reasoning across a breadth of tasks. To address this, we present MultiMedQA, a benchmark combining six existing open question answering datasets spanning professional medical exams, research, and consumer queries; and HealthSearchQA, a new f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.13138","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-12-26T14:28:24Z","cross_cats_sorted":[],"title_canon_sha256":"1df590ae3c941e20696ca9fcfea8716917243676cb2a98fef7067f3fe13fd5c2","abstract_canon_sha256":"0eed6a52166fe2947679cb6443234ddf0d168150de731c28ac9c663ab46c1c3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:28:17.168030Z","signature_b64":"FH1Z4ASLWK1IkeeCz7UEgtvc+nM7VIXDHlZX0Kp1tw03MZ0lnd7MqNj90xwv1QcsmJnknN6GXa5kdBbVA7nbDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93de33ec4e4d6579297e11999e08739ea195d37da1d5ae362d6f158707ab7648","last_reissued_at":"2026-07-05T05:28:17.167555Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:28:17.167555Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models Encode Clinical Knowledge","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aakanksha Chowdhery, Ajay Tanwani, Alan Karthikesalingam, Alvin Rajkomar, Blaise Aguera y Arcas, Chris Kelly, Christopher Semturs, Dale Webster, Greg S. Corrado, Heather Cole-Lewis, Hyung Won Chung, Jason Wei, Joelle Barral, Juraj Gottweis, Karan Singhal, Katherine Chou, Martin Seneviratne, Nathaneal Scharli, Nathan Scales, Nenad Tomasev, Paul Gamble, Perry Payne, Philip Mansfield, Shekoofeh Azizi, S. Sara Mahdavi, Stephen Pfohl, Tao Tu, Vivek Natarajan, Yossi Matias, Yun Liu","submitted_at":"2022-12-26T14:28:24Z","abstract_excerpt":"Large language models (LLMs) have demonstrated impressive capabilities in natural language understanding and generation, but the quality bar for medical and clinical applications is high. Today, attempts to assess models' clinical knowledge typically rely on automated evaluations on limited benchmarks. There is no standard to evaluate model predictions and reasoning across a breadth of tasks. To address this, we present MultiMedQA, a benchmark combining six existing open question answering datasets spanning professional medical exams, research, and consumer queries; and HealthSearchQA, a new f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.13138","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.13138/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.13138","created_at":"2026-07-05T05:28:17.167613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.13138v1","created_at":"2026-07-05T05:28:17.167613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.13138","created_at":"2026-07-05T05:28:17.167613+00:00"},{"alias_kind":"pith_short_12","alias_value":"SPPDH3COJVSX","created_at":"2026-07-05T05:28:17.167613+00:00"},{"alias_kind":"pith_short_16","alias_value":"SPPDH3COJVSXSKL6","created_at":"2026-07-05T05:28:17.167613+00:00"},{"alias_kind":"pith_short_8","alias_value":"SPPDH3CO","created_at":"2026-07-05T05:28:17.167613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19509","citing_title":"LLM Doesn't Know What It Doesn't Know: Detecting Epistemic Blind Spots via Cross-Model Attribution Divergence on Clinical Tabular Data","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31432","citing_title":"Clinically Structured Rank-Gated LoRA for Cross-Benchmark Medical Question Answering","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31432","citing_title":"Clinically Structured Rank-Gated LoRA for Cross-Benchmark Medical Question Answering","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29375","citing_title":"TriageRA-CCF: Source-Side Clinical Confidence and Coverage Signals for Adaptive Rank Budgeting in Medical LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28526","citing_title":"A French OSCE Dialogue Dataset and Controllable Virtual Patient System for Clinical Training","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2301.13688","citing_title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2305.09617","citing_title":"Towards Expert-Level Medical Question Answering with Large Language Models","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":273,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23304","citing_title":"MedGemma vs GPT-4: Open-Source and Proprietary Zero-shot Medical Disease Classification from Images","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2305.10415","citing_title":"PMC-VQA: Visual Instruction Tuning for Medical Visual Question Answering","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2303.13375","citing_title":"Capabilities of GPT-4 on Medical Challenge Problems","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2401.18059","citing_title":"RAPTOR: Recursive Abstractive Processing for Tree-Organized Retrieval","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15156","citing_title":"MeMo: Memory as a Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17564","citing_title":"BloombergGPT: A Large Language Model for Finance","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2310.05737","citing_title":"Language Model Beats Diffusion -- Tokenizer is Key to Visual Generation","ref_index":254,"is_internal_anchor":false},{"citing_arxiv_id":"2402.06196","citing_title":"Large Language Models: A Survey","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10079","citing_title":"Why Supervised Fine-Tuning Fails to Learn: A Systematic Study of Incomplete Learning in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08212","citing_title":"Vision-Language Foundation Models for Comprehensive Automated Pavement Condition Assessment","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05435","citing_title":"CareTransition-Audit: A Benchmark to Audit Discharge Summaries for Efficient Care Transitions","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01048","citing_title":"Compared to What? Baselines and Metrics for Counterfactual Prompting","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03916","citing_title":"Atomic Fact-Checking Increases Clinician Trust in Large Language Model Recommendations for Oncology Decision Support: A Randomized Controlled Trial","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2","json":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2.json","graph_json":"https://pith.science/api/pith-number/SPPDH3COJVSXSKL6CGMZ4CDTT2/graph.json","events_json":"https://pith.science/api/pith-number/SPPDH3COJVSXSKL6CGMZ4CDTT2/events.json","paper":"https://pith.science/paper/SPPDH3CO"},"agent_actions":{"view_html":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2","download_json":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2.json","view_paper":"https://pith.science/paper/SPPDH3CO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.13138&json=true","fetch_graph":"https://pith.science/api/pith-number/SPPDH3COJVSXSKL6CGMZ4CDTT2/graph.json","fetch_events":"https://pith.science/api/pith-number/SPPDH3COJVSXSKL6CGMZ4CDTT2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2/action/storage_attestation","attest_author":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2/action/author_attestation","sign_citation":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2/action/citation_signature","submit_replication":"https://pith.science/pith/SPPDH3COJVSXSKL6CGMZ4CDTT2/action/replication_record"}},"created_at":"2026-07-05T05:28:17.167613+00:00","updated_at":"2026-07-05T05:28:17.167613+00:00"}