{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4XOWRDXH57NEZQ525FCY6LRMEP","short_pith_number":"pith:4XOWRDXH","schema_version":"1.0","canonical_sha256":"e5dd688ee7efda4cc3bae9458f2e2c23c2c88a183e9db03f799488453bd0969d","source":{"kind":"arxiv","id":"2506.18378","version":1},"attestation_state":"computed","paper":{"title":"Taming Vision-Language Models for Medical Image Analysis: A Comprehensive Review","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"eess.IV","authors_text":"Cheng Xu, Haoneng Lin, Jing Qin","submitted_at":"2025-06-23T08:11:24Z","abstract_excerpt":"Modern Vision-Language Models (VLMs) exhibit unprecedented capabilities in cross-modal semantic understanding between visual and textual modalities. Given the intrinsic need for multi-modal integration in clinical applications, VLMs have emerged as a promising solution for a wide range of medical image analysis tasks. However, adapting general-purpose VLMs to medical domain poses numerous challenges, such as large domain gaps, complicated pathological variations, and diversity and uniqueness of different tasks. The central purpose of this review is to systematically summarize recent advances i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.18378","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.IV","submitted_at":"2025-06-23T08:11:24Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"0ed8ecb1e8b210120af897b74501af5f98899c02140037d767d9051dd379bd59","abstract_canon_sha256":"7ae021e5a7cf7f0a5fb96b066d2ae614aedd6b153d5cad7967a9b17feef64a5f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:49.283637Z","signature_b64":"2jyIMc560wijTyS591oJatsRASIRjSIUxKkmPPuSrJGC/YaRMZElff+I0SzfIjgL3GZk1YGs1YVsyie4Ib8RBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5dd688ee7efda4cc3bae9458f2e2c23c2c88a183e9db03f799488453bd0969d","last_reissued_at":"2026-07-05T11:25:49.283159Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:49.283159Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taming Vision-Language Models for Medical Image Analysis: A Comprehensive Review","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"eess.IV","authors_text":"Cheng Xu, Haoneng Lin, Jing Qin","submitted_at":"2025-06-23T08:11:24Z","abstract_excerpt":"Modern Vision-Language Models (VLMs) exhibit unprecedented capabilities in cross-modal semantic understanding between visual and textual modalities. Given the intrinsic need for multi-modal integration in clinical applications, VLMs have emerged as a promising solution for a wide range of medical image analysis tasks. However, adapting general-purpose VLMs to medical domain poses numerous challenges, such as large domain gaps, complicated pathological variations, and diversity and uniqueness of different tasks. The central purpose of this review is to systematically summarize recent advances i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.18378","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.18378/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.18378","created_at":"2026-07-05T11:25:49.283218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.18378v1","created_at":"2026-07-05T11:25:49.283218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.18378","created_at":"2026-07-05T11:25:49.283218+00:00"},{"alias_kind":"pith_short_12","alias_value":"4XOWRDXH57NE","created_at":"2026-07-05T11:25:49.283218+00:00"},{"alias_kind":"pith_short_16","alias_value":"4XOWRDXH57NEZQ52","created_at":"2026-07-05T11:25:49.283218+00:00"},{"alias_kind":"pith_short_8","alias_value":"4XOWRDXH","created_at":"2026-07-05T11:25:49.283218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17140","citing_title":"UCSF-PDGM-VQA: Visual Question Answering dataset for brain tumor MRI interpretation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15736","citing_title":"BiomedAP: A Vision-Informed Dual-Anchor Framework with Gated Cross-Modal Fusion for Robust Medical Vision-Language Adaptation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17140","citing_title":"UCSF-PDGM-VQA: Visual Question Answering dataset for brain tumor MRI interpretation","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP","json":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP.json","graph_json":"https://pith.science/api/pith-number/4XOWRDXH57NEZQ525FCY6LRMEP/graph.json","events_json":"https://pith.science/api/pith-number/4XOWRDXH57NEZQ525FCY6LRMEP/events.json","paper":"https://pith.science/paper/4XOWRDXH"},"agent_actions":{"view_html":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP","download_json":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP.json","view_paper":"https://pith.science/paper/4XOWRDXH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.18378&json=true","fetch_graph":"https://pith.science/api/pith-number/4XOWRDXH57NEZQ525FCY6LRMEP/graph.json","fetch_events":"https://pith.science/api/pith-number/4XOWRDXH57NEZQ525FCY6LRMEP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP/action/storage_attestation","attest_author":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP/action/author_attestation","sign_citation":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP/action/citation_signature","submit_replication":"https://pith.science/pith/4XOWRDXH57NEZQ525FCY6LRMEP/action/replication_record"}},"created_at":"2026-07-05T11:25:49.283218+00:00","updated_at":"2026-07-05T11:25:49.283218+00:00"}