{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QY7X76C6S27AWV5WD2HO3FJYS6","short_pith_number":"pith:QY7X76C6","schema_version":"1.0","canonical_sha256":"863f7ff85e96be0b57b61e8eed953897861ecee3920f01f36363dc74f0877629","source":{"kind":"arxiv","id":"2304.05368","version":3},"attestation_state":"computed","paper":{"title":"Are Large Language Models Ready for Healthcare? A Comparative Study on Clinical Language Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Linda Petzold, Yun Zhao, Yuqing Wang","submitted_at":"2023-04-09T16:31:47Z","abstract_excerpt":"Large language models (LLMs) have made significant progress in various domains, including healthcare. However, the specialized nature of clinical language understanding tasks presents unique challenges and limitations that warrant further investigation. In this study, we conduct a comprehensive evaluation of state-of-the-art LLMs, namely GPT-3.5, GPT-4, and Bard, within the realm of clinical language understanding tasks. These tasks span a diverse range, including named entity recognition, relation extraction, natural language inference, semantic textual similarity, document classification, an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.05368","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-04-09T16:31:47Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ce835c34a412a53939ccb348fb38d6996f08e50c06840b6f0c2612945fba4ca1","abstract_canon_sha256":"3ee91eeec60e0cd78b0f257a841faa5dcfdfac0b350b9c7e9140dbc86a888f5d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:50.279923Z","signature_b64":"jLi3uVQqEM0jE6uPU5/w5xLVWzLLGld2TXjAJLIDLsskuTqe2Dz6A/IdgOBugeUgJEJVYp32h4euUiUCx6z8Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"863f7ff85e96be0b57b61e8eed953897861ecee3920f01f36363dc74f0877629","last_reissued_at":"2026-07-05T06:35:50.279418Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:50.279418Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Large Language Models Ready for Healthcare? A Comparative Study on Clinical Language Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Linda Petzold, Yun Zhao, Yuqing Wang","submitted_at":"2023-04-09T16:31:47Z","abstract_excerpt":"Large language models (LLMs) have made significant progress in various domains, including healthcare. However, the specialized nature of clinical language understanding tasks presents unique challenges and limitations that warrant further investigation. In this study, we conduct a comprehensive evaluation of state-of-the-art LLMs, namely GPT-3.5, GPT-4, and Bard, within the realm of clinical language understanding tasks. These tasks span a diverse range, including named entity recognition, relation extraction, natural language inference, semantic textual similarity, document classification, an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.05368","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.05368/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.05368","created_at":"2026-07-05T06:35:50.279489+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.05368v3","created_at":"2026-07-05T06:35:50.279489+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.05368","created_at":"2026-07-05T06:35:50.279489+00:00"},{"alias_kind":"pith_short_12","alias_value":"QY7X76C6S27A","created_at":"2026-07-05T06:35:50.279489+00:00"},{"alias_kind":"pith_short_16","alias_value":"QY7X76C6S27AWV5W","created_at":"2026-07-05T06:35:50.279489+00:00"},{"alias_kind":"pith_short_8","alias_value":"QY7X76C6","created_at":"2026-07-05T06:35:50.279489+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.07983","citing_title":"Performance and Practical Considerations of Large and Small Language Models in Clinical Decision Support in Rheumatology","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6","json":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6.json","graph_json":"https://pith.science/api/pith-number/QY7X76C6S27AWV5WD2HO3FJYS6/graph.json","events_json":"https://pith.science/api/pith-number/QY7X76C6S27AWV5WD2HO3FJYS6/events.json","paper":"https://pith.science/paper/QY7X76C6"},"agent_actions":{"view_html":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6","download_json":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6.json","view_paper":"https://pith.science/paper/QY7X76C6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.05368&json=true","fetch_graph":"https://pith.science/api/pith-number/QY7X76C6S27AWV5WD2HO3FJYS6/graph.json","fetch_events":"https://pith.science/api/pith-number/QY7X76C6S27AWV5WD2HO3FJYS6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6/action/storage_attestation","attest_author":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6/action/author_attestation","sign_citation":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6/action/citation_signature","submit_replication":"https://pith.science/pith/QY7X76C6S27AWV5WD2HO3FJYS6/action/replication_record"}},"created_at":"2026-07-05T06:35:50.279489+00:00","updated_at":"2026-07-05T06:35:50.279489+00:00"}