{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MSYKUDKSIYIIMQGOXOA36OYX7T","short_pith_number":"pith:MSYKUDKS","schema_version":"1.0","canonical_sha256":"64b0aa0d5246108640cebb81bf3b17fcf5322750dcd659e593c1925bc48d5a1f","source":{"kind":"arxiv","id":"2504.21051","version":1},"attestation_state":"computed","paper":{"title":"Multimodal Large Language Models for Medicine: A Comprehensive Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.LG","authors_text":"Hao Tang, Jiarui Ye","submitted_at":"2025-04-29T03:07:38Z","abstract_excerpt":"MLLMs have recently become a focal point in the field of artificial intelligence research. Building on the strong capabilities of LLMs, MLLMs are adept at addressing complex multi-modal tasks. With the release of GPT-4, MLLMs have gained substantial attention from different domains. Researchers have begun to explore the potential of MLLMs in the medical and healthcare domain. In this paper, we first introduce the background and fundamental concepts related to LLMs and MLLMs, while emphasizing the working principles of MLLMs. Subsequently, we summarize three main directions of application withi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.21051","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-29T03:07:38Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"4c735e20e2c9383c6efb76835ced0817f9fe799b9e328d371ffbc86c528222b1","abstract_canon_sha256":"7fc1babd982253ca8541e0023a71efbf35ad90ad811d6219643d7adf75f0fbfd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:55.832404Z","signature_b64":"cNfGcVuX2jD3EuJuXJIMitZDykyMTQpazUIe5SuXObjNmB95vHFBsW8OM5zAb/UvehQRaB6INxaJBFj0HOI8CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64b0aa0d5246108640cebb81bf3b17fcf5322750dcd659e593c1925bc48d5a1f","last_reissued_at":"2026-07-05T10:55:55.831933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:55.831933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal Large Language Models for Medicine: A Comprehensive Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.LG","authors_text":"Hao Tang, Jiarui Ye","submitted_at":"2025-04-29T03:07:38Z","abstract_excerpt":"MLLMs have recently become a focal point in the field of artificial intelligence research. Building on the strong capabilities of LLMs, MLLMs are adept at addressing complex multi-modal tasks. With the release of GPT-4, MLLMs have gained substantial attention from different domains. Researchers have begun to explore the potential of MLLMs in the medical and healthcare domain. In this paper, we first introduce the background and fundamental concepts related to LLMs and MLLMs, while emphasizing the working principles of MLLMs. Subsequently, we summarize three main directions of application withi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.21051","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.21051/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.21051","created_at":"2026-07-05T10:55:55.831990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.21051v1","created_at":"2026-07-05T10:55:55.831990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.21051","created_at":"2026-07-05T10:55:55.831990+00:00"},{"alias_kind":"pith_short_12","alias_value":"MSYKUDKSIYII","created_at":"2026-07-05T10:55:55.831990+00:00"},{"alias_kind":"pith_short_16","alias_value":"MSYKUDKSIYIIMQGO","created_at":"2026-07-05T10:55:55.831990+00:00"},{"alias_kind":"pith_short_8","alias_value":"MSYKUDKS","created_at":"2026-07-05T10:55:55.831990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26320","citing_title":"MULTISEISMO: A Multimodal Seismic Dataset and Model for Cross-Modal Seismic Understanding","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02001","citing_title":"Generating Findings for Jaw Cysts in Dental Panoramic Radiographs Using a GPT-Based VLM: A Preliminary Study on Building a Two-Stage Self-Correction Loop with Structured Output (SLSO) Framework","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25296","citing_title":"Learning from Medical Entity Trees: An Entity-Centric Medical Data Engineering Framework for MLLMs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08815","citing_title":"Towards Responsible Multimodal Medical Reasoning via Context-Aligned Vision-Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08333","citing_title":"Lost in the Hype: Revealing and Dissecting the Performance Degradation of Medical Multimodal Large Language Models in Image Classification","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T","json":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T.json","graph_json":"https://pith.science/api/pith-number/MSYKUDKSIYIIMQGOXOA36OYX7T/graph.json","events_json":"https://pith.science/api/pith-number/MSYKUDKSIYIIMQGOXOA36OYX7T/events.json","paper":"https://pith.science/paper/MSYKUDKS"},"agent_actions":{"view_html":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T","download_json":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T.json","view_paper":"https://pith.science/paper/MSYKUDKS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.21051&json=true","fetch_graph":"https://pith.science/api/pith-number/MSYKUDKSIYIIMQGOXOA36OYX7T/graph.json","fetch_events":"https://pith.science/api/pith-number/MSYKUDKSIYIIMQGOXOA36OYX7T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T/action/storage_attestation","attest_author":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T/action/author_attestation","sign_citation":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T/action/citation_signature","submit_replication":"https://pith.science/pith/MSYKUDKSIYIIMQGOXOA36OYX7T/action/replication_record"}},"created_at":"2026-07-05T10:55:55.831990+00:00","updated_at":"2026-07-05T10:55:55.831990+00:00"}