{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DW6UURGYYZKA3HEEFBHTTEYQ5O","short_pith_number":"pith:DW6UURGY","schema_version":"1.0","canonical_sha256":"1dbd4a44d8c6540d9c84284f399310eb8995c84d984a362410cfc6b2bb251b5e","source":{"kind":"arxiv","id":"2404.15777","version":4},"attestation_state":"computed","paper":{"title":"A Comprehensive Survey on Evaluating Large Language Model Applications in the Medical Industry","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boyuan Wang, Keke Tang, Meilian Chen, Yining Huang","submitted_at":"2024-04-24T09:55:24Z","abstract_excerpt":"Since the inception of the Transformer architecture in 2017, Large Language Models (LLMs) such as GPT and BERT have evolved significantly, impacting various industries with their advanced capabilities in language understanding and generation. These models have shown potential to transform the medical field, highlighting the necessity for specialized evaluation frameworks to ensure their effective and ethical deployment. This comprehensive survey delineates the extensive application and requisite evaluation of LLMs within healthcare, emphasizing the critical need for empirical validation to ful"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.15777","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-24T09:55:24Z","cross_cats_sorted":[],"title_canon_sha256":"3882f47095e46edc935add3e9017c383e2f13cdfc708dd883082682cbee2a116","abstract_canon_sha256":"21d5d4f38c66ee95dfd02b6f561ca2ae8ef4f621c7ef6f05c82ef457e93a355a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:42.231721Z","signature_b64":"7J0P0JAC1nyVmPrTmHH5NHqqRSses5TL1CaN5yKSIoJUoLw0FIiutnAnu++WJS+8fJ6K8zJ67xUSIEnmnT2aDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1dbd4a44d8c6540d9c84284f399310eb8995c84d984a362410cfc6b2bb251b5e","last_reissued_at":"2026-07-05T08:24:42.231292Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:42.231292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Survey on Evaluating Large Language Model Applications in the Medical Industry","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boyuan Wang, Keke Tang, Meilian Chen, Yining Huang","submitted_at":"2024-04-24T09:55:24Z","abstract_excerpt":"Since the inception of the Transformer architecture in 2017, Large Language Models (LLMs) such as GPT and BERT have evolved significantly, impacting various industries with their advanced capabilities in language understanding and generation. These models have shown potential to transform the medical field, highlighting the necessity for specialized evaluation frameworks to ensure their effective and ethical deployment. This comprehensive survey delineates the extensive application and requisite evaluation of LLMs within healthcare, emphasizing the critical need for empirical validation to ful"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.15777","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.15777/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.15777","created_at":"2026-07-05T08:24:42.231360+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.15777v4","created_at":"2026-07-05T08:24:42.231360+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.15777","created_at":"2026-07-05T08:24:42.231360+00:00"},{"alias_kind":"pith_short_12","alias_value":"DW6UURGYYZKA","created_at":"2026-07-05T08:24:42.231360+00:00"},{"alias_kind":"pith_short_16","alias_value":"DW6UURGYYZKA3HEE","created_at":"2026-07-05T08:24:42.231360+00:00"},{"alias_kind":"pith_short_8","alias_value":"DW6UURGY","created_at":"2026-07-05T08:24:42.231360+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":180,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28526","citing_title":"A French OSCE Dialogue Dataset and Controllable Virtual Patient System for Clinical Training","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O","json":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O.json","graph_json":"https://pith.science/api/pith-number/DW6UURGYYZKA3HEEFBHTTEYQ5O/graph.json","events_json":"https://pith.science/api/pith-number/DW6UURGYYZKA3HEEFBHTTEYQ5O/events.json","paper":"https://pith.science/paper/DW6UURGY"},"agent_actions":{"view_html":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O","download_json":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O.json","view_paper":"https://pith.science/paper/DW6UURGY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.15777&json=true","fetch_graph":"https://pith.science/api/pith-number/DW6UURGYYZKA3HEEFBHTTEYQ5O/graph.json","fetch_events":"https://pith.science/api/pith-number/DW6UURGYYZKA3HEEFBHTTEYQ5O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O/action/storage_attestation","attest_author":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O/action/author_attestation","sign_citation":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O/action/citation_signature","submit_replication":"https://pith.science/pith/DW6UURGYYZKA3HEEFBHTTEYQ5O/action/replication_record"}},"created_at":"2026-07-05T08:24:42.231360+00:00","updated_at":"2026-07-05T08:24:42.231360+00:00"}