{"schema":"pith.reference-change-event.v1","doi":"10.1038/s41586-023-06291-2","canonical_url":"https://pith.science/event/10.1038/s41586-023-06291-2","json_url":"https://pith.science/event/10.1038/s41586-023-06291-2.json","not_a_judgment":"This page records that a citing paper's bibliography includes a work with a published notice. It is not a judgment on the citing paper.","primary":{"event_id":323080,"doi":"10.1038/s41586-023-06291-2","event_type":"correction","event_type_label":"Correction","source":"crossref","source_label":"Crossref","event_date":"2023-07-27","title":"Publisher Correction: Large language models encode clinical knowledge","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","work_arxiv_id":null,"notice_doi":"10.1038/s41586-023-06455-0","flag_count":0,"flags_open":0,"flags_disputed":0,"latest_flag_at":null,"human_href":"/event/10.1038/s41586-023-06291-2","json_href":"/event/10.1038/s41586-023-06291-2.json"},"events":[{"event_id":323080,"doi":"10.1038/s41586-023-06291-2","event_type":"correction","event_type_label":"Correction","source":"crossref","source_label":"Crossref","event_date":"2023-07-27","title":"Publisher Correction: Large language models encode clinical knowledge","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","work_arxiv_id":null,"notice_doi":"10.1038/s41586-023-06455-0","flag_count":0,"flags_open":0,"flags_disputed":0,"latest_flag_at":null,"human_href":"/event/10.1038/s41586-023-06291-2","json_href":"/event/10.1038/s41586-023-06291-2.json"}],"flags":[{"id":4611,"status":"open","status_label":"Open","citing_arxiv_id":"2604.18302","citing_title":"Toward Zero-Egress Psychiatric AI: On-Device LLM Deployment for Privacy-Preserving Mental Health Decision Support","ref_index":62,"evidence_raw":"K. Singhal, et al., Large language models encode clinical knowledge, Nature 620 (2023) 172–180. doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4611","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2604.18302","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4616,"status":"open","status_label":"Open","citing_arxiv_id":"2605.06856","citing_title":"Benchmarked Yet Not Measured -- Generative AI Should be Evaluated Against Real-World Utility","ref_index":256,"evidence_raw":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Sch. Large language models encode clinical knowledge , journal=. 2023 , month=. doi:10.1038/s41586-023-06291-2 , url=","evidence_cleaned":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Sch. Large language models encode clinical knowledge, journal=. 2023, month=. doi:10.1038/s41586-023-06291-2, url=","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4616","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.06856","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4604,"status":"open","status_label":"Open","citing_arxiv_id":"2605.04180","citing_title":"MedFabric and EtHER: A Data-Centric Framework for Word-Level Fabrication Generation and Detection in Medical LLMs","ref_index":29,"evidence_raw":"Singhal, K., etc: Large language models encode clinical knowledge. Nature 620(7972), 172–180 (2023).https://doi.org/10.1038/s41586-023-06291-2 Title Suppressed Due to Excessive Length 13","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4604","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.04180","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4605,"status":"open","status_label":"Open","citing_arxiv_id":"2605.05715","citing_title":"Decodable but Not Corrected by Fixed Residual-Stream Linear Steering: Evidence from Medical LLM Failure Regimes","ref_index":59,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, and 1 others. 2023. https://doi.org/10.1038/s41586-023-06291-2 Large language models encode clinical knowledge . Nature, 620(7972):172--180","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4605","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.05715","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4617,"status":"open","status_label":"Open","citing_arxiv_id":"2605.10286","citing_title":"AgentRx: A Benchmark Study of LLM Agents for Multimodal Clinical Prediction Tasks","ref_index":120,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Schärli, Aakanksha Chowdhery, Philip Mansfield, Dina Demner-Fushman, Blaise Agüera y Arcas, Dale Webster, Greg S. Corrado, Yossi Matias, Katherine Chou, Juraj Gottweis, Nenad Tomasev, Yun Liu, Alvin Rajkomar, Joelle Barral, Christopher Semturs, Alan Karthikesalingam, and Vivek Natarajan. Large language models encode clinical knowledge. Nature, 620 0 (7972): 0 172--180, August 2023. ISSN 1476-4687. doi:10.1038/s41586-023-06291-2. URL https://www.nature.com/articles/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4617","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.10286","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4606,"status":"open","status_label":"Open","citing_arxiv_id":"2604.27470","citing_title":"HealthBench Professional: Evaluating Large Language Models on Real Clinician Chats","ref_index":2,"evidence_raw":"URLhttps://proceedings.mlr.press/v174/pal22a.html. Khaled Saab, Tao Tu, Wei-Hung Weng, Ryutaro Tanno, David Stutz, Ellery Wulczyn, Fan Zhang, Tim Strother, Chunjong Park, Elahe Vedadi, et al. Capabilities of gemini models in medicine.arXiv preprint arXiv:2404.18416, 2024. Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al. Large language models encode clinical knowledge. 20 Nature, 620:172–180, 2023. doi: 10.1038/s41586-023-06291-2. URLhttps://www.nature.com/articles/ s41586-023-06291-2. Karan Singhal, Tao Tu, Juraj Gottweis, Rory Sayres, Ellery Wulczyn, Mohamed Amin, Le Hou, Kevin Clark, Stephen R. Pfohl, Heather Cole-Lewis, et al. Toward expert-level medical question answering with large language models.Nature Medicine, 31:943–950, 2025. doi: 10.1038/s41591-024-03423-7. URL https://www.nature.com/articles/s41591-024-03423-7. Sarvesh Soni, Soumya Gayen, and Dina Demner-Fushman. Overview of the ArchEHR-QA 2025 shared task on grounded question answering from electronic health records. InProceedings of the 24th Workshop on Biomedical Language Processing, pages 396–405, Vienna, Aust","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4606","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2604.27470","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4607,"status":"open","status_label":"Open","citing_arxiv_id":"2605.03476","citing_title":"CuraView: A Multi-Agent Framework for Medical Hallucination Detection with GraphRAG-Enhanced Knowledge Verification","ref_index":3,"evidence_raw":"Large language models (LLMs) have demonstrated strong potential across medical applications, particularly in clinical documentation tasks such as discharge summary generation, diagnostic as- sistance, and radiology report generation [1][2]. Landmark models including Med-PaLM and Med- PaLM 2 achieved 67.6% and 86.5% accuracy on USMLE-style questions in the MedQA benchmark, respectively [3][4], while GPT-4 demonstrated competitive performance on the MultiMedQA bench- mark [5]. Among clinical documentation tasks, discharge summary generation is of particular safety significance: discharge summaries serve as the primary record guiding post-discharge medication, follow-up care, and inter-provider communication, and errors introduced at this stage propagate","evidence_cleaned":null,"evidence_source_label":"citation context","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4607","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.03476","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4608,"status":"open","status_label":"Open","citing_arxiv_id":"2605.20591","citing_title":"Do No Harm? Hallucination and Actor-Level Abuse in Web-Deployed Medical Large Language Models","ref_index":18,"evidence_raw":"S. Karan, A. Shekoofeh, T. Tao, M. S. Sara, W. Jason, W. C. Hyung, S. Nathan, T. Ajay, C.-L. Heather, P. Stephen, P. Perry, S. Martin, G. Paul, K. Chris, B. Abubakr, S. Nathanael, C. Aakanksha, M. Philip, D.-F. Dina, A. y. A. Blaise, W. Dale, S. C. Greg, M. Yossi, C. Katherine, G. Juraj, T. Nenad, L. Yun, R. Alvin, B. Joelle, S. Christopher, K. Alan, and N. Vivek, “Large language models encode clinical knowledge,”Nature, vol. 620, no. 1, p. 172–180, 2023. [Online]. Available: https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4608","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.20591","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4609,"status":"open","status_label":"Open","citing_arxiv_id":"2605.04085","citing_title":"Evaluating Patient Safety Risks in Generative AI: Development and Validation of a FMECA Framework for Generated Clinical Content","ref_index":4,"evidence_raw":"[2] Feblowitz JC, Wright A, Singh H, Samal L, Sittig DF. Summarization of clinical information: A conceptual model. J Biomed Inform 2011;44:688-99. https://doi.org/10.1016/j.jbi.2011.03.008. [3] Thirunavukarasu AJ, Ting DSJ, Elangovan K, Gutierrez L, Tan TF, Ting DSW. Large language models in medicine. Nat Med 2023;29:1930-40. https://doi.org/10.1038/s41591-023-02448-8. [4] Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large language models encode clinical knowledge. Nature 2023;620:172-80. https://doi.org/10.1038/s41586-023-06291-2. [5] Clusmann J, Kolbinger FR, Muti HS, Carrero ZI, Eckardt J -N, Laleh NG, et al. The future landscape of large language models in medicine. Commun Med 2023;3:141. https://doi.","evidence_cleaned":"[2] Feblowitz JC, Wright A, Singh H, Samal L, Sittig DF. Summarization of clinical information: A conceptual model. J Biomed Inform 2011;44:688-99. https://doi.org/10.1016/j.jbi.2011.03.008. [3] Thirunavukarasu AJ, Ting DSJ, Elangovan K, Gutierrez L, Tan TF, Ting DSW. Large language models in medicine. Nat Med 2023;29:1930-40. https://doi.org/10.1038/s41591-023-02448-8. [4] Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large language models encode clinical knowledge. Nature 2023;620:172-80. https://doi.org/10.1038/s41586-023-06291-2. [5] Clusmann J, Kolbinger FR, Muti HS, Carrero ZI, Eckardt J -N, Laleh NG, et al. The future landscape of large language models in medicine. Commun Med 2023;3:141. https://doi","evidence_source_label":"citation context","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4609","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.04085","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4610,"status":"open","status_label":"Open","citing_arxiv_id":"2605.01664","citing_title":"A Hybrid Retrieval and Reranking Framework for Evidence-Grounded Retrieval-Augmented Generation","ref_index":30,"evidence_raw":"K. Singhal, S. Azizi, T. Tu, S. S. Mahdavi, J. Wei, H. W. Chung, N. Scales, A. Tanwani, H. Cole-Lewis, S. Pfohl, P. Payne, M. Seneviratne, P. Gamble, C. Kelly, A. Babiker, N. Sch ¨arli, A. Chowdhery, P. Mansfield, B. Demner-Fushman, F. Ag ¨uera y Arcas, D. Webster, G. S. Corrado, Y . Matias, K. Chou, J. Gottweis, N. Tomasev, Y . Liu, A. Rajkomar, J. Barral, C. Semturs, A. Karthikesalingam, and V . Natarajan, “Large language models encode clinical knowledge,”Nature, vol. 620, pp. 172– 180, 2023. doi: 10.1038/s41586-023-06291-2. [Online]. Available: https: //www.nature.com/articles/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4610","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.01664","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4612,"status":"open","status_label":"Open","citing_arxiv_id":"2604.07813","citing_title":"Agentivism: a learning theory for the age of artificial intelligence","ref_index":57,"evidence_raw":"AI: A case study of a generative AI-based knowledge-building learning companion for teachers. British Journal of Educational Technology. 2026;https://doi.org/10. 1111/bjet.70013. [56] Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large Language Models Encode Clinical Knowledge. Nature. 2023;620(7972):172-180. https:// doi.org/10.1038/s41586-023-06291-2. [57] Kraemer MUG, et al. Artificial Intelligence for Modelling Infectious Dis- ease Epidemics. Nature. 2025;638(8051):623-635. https://doi.org/10.1038/ s41586-024-08564-w. [58] Rao V, et al. Multimodal Generative AI for Medical Image Interpretation. Nature. 2025;639(8056):888-896. https://doi.org/10.1038/s41586-025-08675-y. [59] Celik I, Kontkanen S, Laru J, Dalyanci AA.","evidence_cleaned":"AI: A case study of a generative AI-based knowledge-building learning companion for teachers. British Journal of Educational Technology. 2026;https://doi.org/10. 1111/bjet.70013. [56] Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large Language Models Encode Clinical Knowledge. Nature. 2023;620(7972):172-180. https:// doi.org/10.1038/s41586-023-06291-2. [57] Kraemer MUG, et al. Artificial Intelligence for Modelling Infectious Dis- ease Epidemics. Nature. 2025;638(8051):623-635. https://doi.org/10.1038/ s41586-024-08564-w. [58] Rao V, et al. Multimodal Generative AI for Medical Image Interpretation. Nature. 2025;639(8056):888-896. https://doi.org/10.1038/s41586-025-08675-y. [59] Celik I, Kontkanen S, Laru J, Dalyanci AA","evidence_source_label":"citation context","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4612","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2604.07813","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4613,"status":"open","status_label":"Open","citing_arxiv_id":"2605.06856","citing_title":"Benchmarked Yet Not Measured -- Generative AI Should be Evaluated Against Real-World Utility","ref_index":256,"evidence_raw":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Sch. Large language models encode clinical knowledge , journal=. 2023 , month=. doi:10.1038/s41586-023-06291-2 , url=","evidence_cleaned":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Sch. Large language models encode clinical knowledge, journal=. 2023, month=. doi:10.1038/s41586-023-06291-2, url=","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4613","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.06856","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4614,"status":"open","status_label":"Open","citing_arxiv_id":"2605.08257","citing_title":"Research on Security Enhancement Methods for Adversarial Robust Large Language Model Intelligent Agents for Medical Decision-Making Tasks","ref_index":1,"evidence_raw":"Singhal, K., Azizi, S., Tu, T., et al. (2023). Large language models encode clinical knowledge. Nature, 620, 172–180. https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4614","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.08257","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4615,"status":"open","status_label":"Open","citing_arxiv_id":"2605.10025","citing_title":"Medical Incident Causal Factors and Preventive Measures Generation Using Tag-based Example Selection in Few-shot Learning","ref_index":28,"evidence_raw":"K. Singhal, S. Azizi, T. Tu, S. S. Mahdavi, J. Wei, H. W. Chung, N. Scales, A. Tanwani, H. Cole-Lewis, S. Pfohl, P. Payne, M. Seneviratne, P. Gamble, C. Kelly, A. Babiker, N. Sch ¨arli, A. Chowdhery, P. Mansfield, D. Demner-Fushman, B. Ag ¨uera y Arcas, D. Webster, G. S. Corrado, Y . Matias, K. Chou, J. Gottweis, N. Tomasev, Y . Liu, A. Rajkomar, J. Barral, C. Semturs, A. Karthikesalingam, and V . Natarajan, “Large language models encode clinical knowledge,” vol. 620, no. 7972, pp. 172–180. [Online]. Available: https: //doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4615","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.10025","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4619,"status":"open","status_label":"Open","citing_arxiv_id":"2603.18294","citing_title":"The Validity Gap in Health AI Evaluation: A Cross-Sectional Analysis of Benchmark Composition","ref_index":10,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Senevi- ratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Sch¨ arli, Aakanksha Chowdh- ery, Philip Mansfield, Dina Demner-Fushman, Blaise Ag¨ uera y Arcas, Dale Webster, Greg S. Corrado, Yossi Matias, Katherine Chou, Juraj Gottweis, Nenad Tomasev, Yun Liu, Alvin Rajkomar, Joelle Barral, Christopher Semturs, Alan Karthikesalingam, and Vivek Natarajan. Large language models encode clinical knowledge.Nature, 620(7972):172–180, August 2023. ISSN 1476-4687. doi: 10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4619","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2603.18294","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4620,"status":"open","status_label":"Open","citing_arxiv_id":"2604.09550","citing_title":"HyEm: Query-Adaptive Hyperbolic Retrieval for Biomedical Ontologies via Euclidean Vector Indexing","ref_index":20,"evidence_raw":"K. Singhal,S. Azizi,T. Tu,S. S. Mahdavi,J. Wei,H. W. Chung,N. Scales, A. Tanwani,H. Cole-Lewis,S. Pfohl,P. Payne,M. Seneviratne,P. Gamble, C. Kelly, A. Babiker, N. Schärli, A. Chowdhery, P. Mansfield, D. Demner- Fushman, B. Agüera y Arcas, D. Webster, G. S. Corrado, Y. Matias, K. Chou, J. Gottweis, N. Tomasev, Y. Liu, A. Rajkomar, J. Barral, C. Semturs, A. Karthikesalingam, V. Natarajan, Large language models encode clinical knowledge, Nature 620 (7972) (2023) 172–180.doi: 10.1038/s41586-023-06291-2. URLhttps://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4620","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2604.09550","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4621,"status":"open","status_label":"Open","citing_arxiv_id":"2507.12261","citing_title":"Infherno: End-to-end Agent-based FHIR Resource Synthesis from Free-form Clinical Notes","ref_index":27,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al. 2023. https://doi.org/10.1038/s41586-023-06291-2 Large language models encode clinical knowledge . Nature, 620(7972):172--180","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4621","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2507.12261","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4622,"status":"open","status_label":"Open","citing_arxiv_id":"2509.24186","citing_title":"Measuring Competency, Not Performance: Item-Aware Evaluation Across Medical Benchmarks","ref_index":27,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al. Large language models encode clinical knowledge. Nature, 620 0 (7972): 0 172--180, 2023. doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4622","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2509.24186","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4623,"status":"open","status_label":"Open","citing_arxiv_id":"2605.17679","citing_title":"PULSE: Agentic Investigation with Passive Sensing for Proactive Intervention in Cancer Survivorship","ref_index":59,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al . 2023. Large Language Models Encode Clinical Knowledge.Nature620, 7972 (2023), 172–180. doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4623","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.17679","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4624,"status":"open","status_label":"Open","citing_arxiv_id":"2605.18768","citing_title":"ClinQueryAgent: A Conversational Agent for Population Health Management","ref_index":89,"evidence_raw":"Large language models encode clinical knowledge , volume =. Nature , author =. 2023 , pages =. doi:10.1038/s41586-023-06291-2 , abstract =","evidence_cleaned":"Large language models encode clinical knowledge, volume =. Nature, author =. 2023, pages =. doi:10.1038/s41586-023-06291-2, abstract =","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4624","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.18768","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4625,"status":"open","status_label":"Open","citing_arxiv_id":"2605.22080","citing_title":"JMed48k: A Multi-Profession Japanese Medical Licensing Benchmark for Vision-Language Model Evaluation","ref_index":48,"evidence_raw":"K. Singhal, S. Azizi, T. Tu, S. S. Mahdavi, J. Wei, H. W. Chung, N. Scales, A. Tanwani, H. Cole-Lewis, S. Pfohl, P. Payne, M. Seneviratne, P. Gamble, C. Kelly, A. Babiker, N. Schärli, A. Chowdhery, P. Mansfield, D. Demner-Fushman, B. Agüera y Arcas, D. Webster, G. S. Cor- rado, Y . Matias, K. Chou, J. Gottweis, N. Tomasev, Y . Liu, A. Rajkomar, J. Barral, C. Semturs, A. Karthikesalingam, and V . Natarajan. Large language models encode clinical knowledge. Nature, 620(7972):172–180, 2023. doi: 10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4625","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.22080","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4628,"status":"open","status_label":"Open","citing_arxiv_id":"2606.24579","citing_title":"Cross-Lingual Exploration for Parametric Knowledge","ref_index":58,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, and 1 others. 2023. https://doi.org/10.1038/s41586-023-06291-2 Large language models encode clinical knowledge . Nature, 620(7972):172--180","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4628","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.24579","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4629,"status":"open","status_label":"Open","citing_arxiv_id":"2606.20164","citing_title":"MedRLM: Recursive Multimodal Health Intelligence for Long-Context Clinical Reasoning, Sensor-Guided Screening, Evidence-Grounded Decision Support, and Community-to-Tertiary Referral Optimization","ref_index":8,"evidence_raw":"K. Singhalet al., “Large Language Models Encode Clinical Knowledge,”Nature,vol.620,pp.172–180,2023.[Online].Avail- able: https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4629","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.20164","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4630,"status":"open","status_label":"Open","citing_arxiv_id":"2606.26519","citing_title":"What the LLM Should Not Say: Boundary-Aware Context Grounding for A Seven-Channel EEG Agent","ref_index":15,"evidence_raw":"Singhal, K., et al. (2023). Large language models encode clinical knowledge.Nature, 620, 172–180.https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4630","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.26519","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4631,"status":"open","status_label":"Open","citing_arxiv_id":"2606.21517","citing_title":"MedHal-Loc: Are \"Explainable-by-Architecture\" Medical Hallucination Detectors Faithful Localizers? A Localization Benchmark","ref_index":1,"evidence_raw":"K. Singhal, S. Azizi, T. Tu, S. S. Mahdavi, J. Wei, H. W. Chung, et al., Large language models encode clinical knowledge, Nature 620 (7972) (2023) 172–180.doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4631","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.21517","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4626,"status":"open","status_label":"Open","citing_arxiv_id":"2406.04244","citing_title":"Benchmark Data Contamination of Large Language Models: A Survey","ref_index":136,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al. 2023. Large language models encode clinical knowledge. Nature 620, 7972 (2023), 172–180. https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4626","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2406.04244","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4627,"status":"open","status_label":"Open","citing_arxiv_id":"2604.07813","citing_title":"Agentivism: a learning theory for the age of artificial intelligence","ref_index":57,"evidence_raw":"Singhal K, Azizi S, Tu T, Mahdavi SS, Wei J, Chung HW, et al. Large Language Models Encode Clinical Knowledge. Nature. 2023;620(7972):172–180. https:// doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4627","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2604.07813","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4632,"status":"open","status_label":"Open","citing_arxiv_id":"2606.18596","citing_title":"Better Adherence, Richer Context: A Field Evaluation of LLM-Powered Conversational Voice Diaries for Sleep","ref_index":62,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Nathaneal Scharli, Aakanksha Chowdhery, Philip Mansfield, Blaise Agüera y Arcas, Dale Webster, Greg S. Corrado, Yossi Matias, Katherine Chou, Juraj Gottweis, Nenad Tomasev, Yun Liu, Alvin Rajkomar, Joelle Barral, Christopher Semturs, Alan Karthikesalingam, and Vivek Natarajan. 2023. Large Language Models Encode Clinical Knowledge.Nature620 (2023), 172–180. https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4632","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.18596","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4633,"status":"open","status_label":"Open","citing_arxiv_id":"2606.12291","citing_title":"Measuring Epistemic Resilience of LLMs Under Misleading Medical Context","ref_index":40,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, et al. Large language models encode clinical knowledge.Nature, 620(7972):172–180, 2023. doi: 10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4633","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.12291","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4634,"status":"open","status_label":"Open","citing_arxiv_id":"2606.12702","citing_title":"Deployment-Centered Evaluation: Predicting Query-Level Rejection Risk in a Clinical LLM System","ref_index":33,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Schärli, Aakanksha Chowdhery, Philip Mansfield, Dina Demner-Fushman, Blaise Agüera y Arcas, Dale Web- ster, Greg S. Corrado, Yossi Matias, Katherine Chou, Juraj Gottweis, Nenad Tomasev, Yun Liu, Alvin Rajkomar, Joelle Barral, Christopher Semturs, Alan Karthikesalingam, and Vivek Natarajan. Large language models encode clinical knowledge.Nature, 620(7972): 172–180, July 2023. ISSN 1476-4687. doi: 10.1038/s41586-023-06291-2. URL http: //dx.doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4634","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.12702","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4635,"status":"open","status_label":"Open","citing_arxiv_id":"2606.12578","citing_title":"MARD: Mirror-Augmented Reasoning Distillation for Mechanism-Level Drug-Drug Interaction Prediction","ref_index":29,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Sch\\\" a rli, Aakanksha Chowdhery, Philip Mansfield, Dina Demner-Fushman, and 13 others. 2023. https://doi.org/10.1038/s41586-023-06291-2 Large language models encode clinical knowledge . Nature, 620(7972):172--180","evidence_cleaned":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Sch a rli, Aakanksha Chowdhery, Philip Mansfield, Dina Demner-Fushman, and 13 others. 2023. https://doi.org/10.1038/s41586-023-06291-2 Large language models encode clinical knowledge . Nature, 620(7972):172--180","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4635","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.12578","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4636,"status":"open","status_label":"Open","citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":196,"evidence_raw":"Large language models encode clinical knowledge , author=. Nature , volume=. 2023 , month=. doi:10.1038/s41586-023-06291-2 , url=","evidence_cleaned":"Large language models encode clinical knowledge, author=. Nature, volume=. 2023, month=. doi:10.1038/s41586-023-06291-2, url=","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4636","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.11740","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4637,"status":"open","status_label":"Open","citing_arxiv_id":"2606.04127","citing_title":"When Retrieval Doesn't Help: A Large-Scale Study of Biomedical RAG","ref_index":24,"evidence_raw":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Schärli, Nathanael and Chowdhery, Aakanksha and Mansfield, Philip and Demner-Fushman, Dina and Agüera y Arcas, Blaise and Webster, Dale and Corrado, Greg S. and Matias, Yossi and Chou, Katherine and Gottweis, Juraj and Tomasev, Nenad and Liu, Yun and Rajkomar, Alvin and Barral, Joelle and Semturs, Christopher and Karthikesalingam, Alan and Natarajan, Vivek , title =. Nature , volume =. 2023 , type =. doi:10.1038/s41586-023-06291-2 , url =","evidence_cleaned":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Schärli, Nathanael and Chowdhery, Aakanksha and Mansfield, Philip and Demner-Fushman, Dina and Agüera y Arcas, Blaise and Webster, Dale and Corrado, Greg S. and Matias, Yossi and Chou, Katherine and Gottweis, Juraj and Tomasev, Nenad and Liu, Yun and Rajkomar, Alvin and Barral, Joelle and Semturs, Christopher and Karthikesalingam, Alan and Natarajan, Vivek, title =. Nature, volume =. 2023, type =. doi:10.1038/s41586-023-06291-2, url =","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4637","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.04127","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4638,"status":"open","status_label":"Open","citing_arxiv_id":"2606.01961","citing_title":"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models","ref_index":68,"evidence_raw":"Karan Singhal, Shekoofeh Azizi Tu, Julia Gottweis, Rory Sayres, Ellery Wulczyn, Le Hou, Peter Schuh, Karan Sareen, David Winer, Denny Wilson, et al. Large language models encode clinical knowledge.Nature, 620:172–180, 2023. doi: 10.1038/s41586-023-06291-2. URL https://www.nature.com/articles/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4638","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.01961","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4639,"status":"open","status_label":"Open","citing_arxiv_id":"2606.26519","citing_title":"What the LLM Should Not Say: Boundary-Aware Context Grounding for A Seven-Channel EEG Agent","ref_index":15,"evidence_raw":"Singhal, K., et al. (2023). Large language models encode clinical knowledge.Nature, 620, 172–180.https://doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4639","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.26519","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4640,"status":"open","status_label":"Open","citing_arxiv_id":"2605.29960","citing_title":"Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction","ref_index":36,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Schärli, Aakanksha Chowdhery, Philip Mansfield, Dina Demner-Fushman, Blaise Agüera Y Arcas, Dale Webster, Greg S. Corrado, Yossi Matias, Katherine Chou, Juraj Gottweis, Nenad Tomasev, Yun Liu, Alvin Rajkomar, Joelle Barral, Christo- pher Semturs, Alan Karthikesalingam, and Vivek Natarajan. 2023. Large Lan- guage Models Encode Clinical Knowledge.Nature620, 7972 (2023), 172–180. doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4640","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.29960","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4641,"status":"open","status_label":"Open","citing_arxiv_id":"2605.27873","citing_title":"AIBuildAI-2: A Knowledge-Enhanced Agent for Automatically Building AI Models","ref_index":4,"evidence_raw":"doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4641","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.27873","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4642,"status":"open","status_label":"Open","citing_arxiv_id":"2605.24573","citing_title":"AstroMind: A High-Fidelity Benchmark for Spacecraft Behavior Reasoning Based on Large Language Models","ref_index":22,"evidence_raw":"K. Singhal, S. Azizi, T. Tu, S. S. Mahdavi, J. Wei, H. W. Chung, N. Scales, A. Tanwani, H. Cole-Lewis, S. Pfohlet al., “Large language models encode clinical knowledge,”Nature, vol. 620, no. 7972, pp. 172–180, 2023. [Online]. Available: https: //doi.org/10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4642","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.24573","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4643,"status":"open","status_label":"Open","citing_arxiv_id":"2605.22080","citing_title":"JMed48k: A Multi-Profession Japanese Medical Licensing Benchmark for Vision-Language Model Evaluation","ref_index":48,"evidence_raw":"K. Singhal, S. Azizi, T. Tu, S. S. Mahdavi, J. Wei, H. W. Chung, N. Scales, A. Tanwani, H. Cole-Lewis, S. Pfohl, P. Payne, M. Seneviratne, P. Gamble, C. Kelly, A. Babiker, N. Schärli, A. Chowdhery, P. Mansfield, D. Demner-Fushman, B. Agüera y Arcas, D. Webster, G. S. Cor- rado, Y . Matias, K. Chou, J. Gottweis, N. Tomasev, Y . Liu, A. Rajkomar, J. Barral, C. Semturs, A. Karthikesalingam, and V . Natarajan. Large language models encode clinical knowledge. Nature, 620(7972):172–180, 2023. doi: 10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4643","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.22080","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4644,"status":"open","status_label":"Open","citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":286,"evidence_raw":"Large language models encode clinical knowledge , volume =. Nature , author =. 2023 , keywords =. doi:10.1038/s41586-023-06291-2 , abstract =","evidence_cleaned":"Large language models encode clinical knowledge, volume =. Nature, author =. 2023, keywords =. doi:10.1038/s41586-023-06291-2, abstract =","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4644","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2606.11219","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4645,"status":"open","status_label":"Open","citing_arxiv_id":"2607.06802","citing_title":"A Multi-Analyst LLM Pipeline for Auditable Rule Discovery Across 68 Public Physiological Corpora","ref_index":17,"evidence_raw":"K. Singhal, S. Azizi, T. Tu,et al., “Large lan- guage models encode clinical knowledge,”Nature, vol. 620, no. 7972, pp. 172–180, 2023, PMID: 37438534, doi: 10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4645","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2607.06802","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4646,"status":"open","status_label":"Open","citing_arxiv_id":"2607.00008","citing_title":"SchemaRAG: Dynamic Large Schema Reduction for LLM-driven Structured Information Extraction","ref_index":26,"evidence_raw":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Sch. Large language models encode clinical knowledge , journal=. 2023 , month=. doi:10.1038/s41586-023-06291-2 , url=","evidence_cleaned":"Singhal, Karan and Azizi, Shekoofeh and Tu, Tao and Mahdavi, S. Sara and Wei, Jason and Chung, Hyung Won and Scales, Nathan and Tanwani, Ajay and Cole-Lewis, Heather and Pfohl, Stephen and Payne, Perry and Seneviratne, Martin and Gamble, Paul and Kelly, Chris and Babiker, Abubakr and Sch. Large language models encode clinical knowledge, journal=. 2023, month=. doi:10.1038/s41586-023-06291-2, url=","evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4646","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2607.00008","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4647,"status":"open","status_label":"Open","citing_arxiv_id":"2607.01440","citing_title":"FaithMed: Training LLMs For Faithful Evidence-Based Medical Reasoning","ref_index":72,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, Perry Payne, Martin Seneviratne, Paul Gamble, Chris Kelly, Abubakr Babiker, Nathanael Schärli, Aakanksha Chowdhery, Philip Mansfield, Dina Demner-Fushman, Blaise Agüera y Arcas, Dale Webster, Greg S. Corrado, Yossi Matias, Katherine Chou, Juraj Gottweis, Nenad Tomašev, Yun Liu, Alvin Rajkomar, Joelle Barral, Christopher Semturs, Alan Karthikesalingam, and Vivek Natarajan. 2023 a . https://doi.org/10.1038/s41586-023-06291-2 Large language models encode clinical knowledge . Nature, 620(7972):172--180","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4647","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2607.01440","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4648,"status":"open","status_label":"Open","citing_arxiv_id":"2607.05311","citing_title":"Deep Learning for Semen Analysis in Male Infertility: Computer Vision, Multimodal Fusion, and Clinical Translation","ref_index":99,"evidence_raw":"Singhal, K., Azizi, S., Tu, T., Mahdavi, S.S., Wei, J., Chung, H.W., Scales, N., Tanwani, A., Cole-Lewis, H., Pfohl, S., et al., 2023a. Large language models encode clinical knowledge. Nature 620, 172–180. doi:10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4648","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2607.05311","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null},{"id":4618,"status":"open","status_label":"Open","citing_arxiv_id":"2605.11143","citing_title":"ClinicalBench: Stress-Testing Assertion-Aware Retrieval for Cross-Admission Clinical QA on MIMIC-IV","ref_index":1,"evidence_raw":"Karan Singhal, Shekoofeh Azizi, Tao Tu, S. Sara Mahdavi, Jason Wei, Hyung Won Chung, Nathan Scales, Ajay Tanwani, Heather Cole-Lewis, Stephen Pfohl, et al. Towards expert-level medical question answering with large language models.Nature, 620: 399–404, 2023. doi: 10.1038/s41586-023-06291-2","evidence_cleaned":null,"evidence_source_label":"bibliography line","event_type":"correction","event_type_label":"Correction","source_label":"Crossref","event_date":"2023-07-27","work_title":"Large lan- guage models encode clinical knowledge,","work_doi":"10.1038/s41586-023-06291-2","event_doi":"10.1038/s41586-023-06291-2","flag_href":"/flags/4618","event_href":"/event/10.1038/s41586-023-06291-2","paper_href":"/paper/2605.11143","created_at":"2026-07-11T03:19:08.018010Z","dispute_note":null,"disputed_at":null,"disputed_by":null}],"flag_count":45,"flags_open":45,"flags_disputed":0,"desk_url":"https://pith.science/flags"}