{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OHD3BRVLE4QAONNTHEIB7HSNPG","short_pith_number":"pith:OHD3BRVL","schema_version":"1.0","canonical_sha256":"71c7b0c6ab27200735b339101f9e4d79ad2bbd3ee971533abba0987841d318cb","source":{"kind":"arxiv","id":"2403.02469","version":2},"attestation_state":"computed","paper":{"title":"Vision-Language Models for Medical Report Generation and Visual Question Answering: A Review","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Ghulam Rasool, Iryna Hartsock","submitted_at":"2024-03-04T20:29:51Z","abstract_excerpt":"Medical vision-language models (VLMs) combine computer vision (CV) and natural language processing (NLP) to analyze visual and textual medical data. Our paper reviews recent advancements in developing VLMs specialized for healthcare, focusing on models designed for medical report generation and visual question answering (VQA). We provide background on NLP and CV, explaining how techniques from both fields are integrated into VLMs to enable learning from multimodal data. Key areas we address include the exploration of medical vision-language datasets, in-depth analyses of architectures and pre-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.02469","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-04T20:29:51Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"938d5b822cf075089cbaa5bbd0bdb132ca9fe83267fc5e4000e88d3a0157eb94","abstract_canon_sha256":"0df5dd1931347e164e3d2db603ccbf19b90c4ef705ba2171bad83193e58a7125"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:32.621003Z","signature_b64":"+x10LJc2DJRPT3iSTdzaarnAmzLGQLH4FhY/Dn51qeFtcb8Es/1+u3Hgv7BuJv/wyFBYs0iE8XDnzJch//2xBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71c7b0c6ab27200735b339101f9e4d79ad2bbd3ee971533abba0987841d318cb","last_reissued_at":"2026-07-05T09:41:32.620561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:32.620561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision-Language Models for Medical Report Generation and Visual Question Answering: A Review","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Ghulam Rasool, Iryna Hartsock","submitted_at":"2024-03-04T20:29:51Z","abstract_excerpt":"Medical vision-language models (VLMs) combine computer vision (CV) and natural language processing (NLP) to analyze visual and textual medical data. Our paper reviews recent advancements in developing VLMs specialized for healthcare, focusing on models designed for medical report generation and visual question answering (VQA). We provide background on NLP and CV, explaining how techniques from both fields are integrated into VLMs to enable learning from multimodal data. Key areas we address include the exploration of medical vision-language datasets, in-depth analyses of architectures and pre-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.02469","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.02469/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.02469","created_at":"2026-07-05T09:41:32.620617+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.02469v2","created_at":"2026-07-05T09:41:32.620617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.02469","created_at":"2026-07-05T09:41:32.620617+00:00"},{"alias_kind":"pith_short_12","alias_value":"OHD3BRVLE4QA","created_at":"2026-07-05T09:41:32.620617+00:00"},{"alias_kind":"pith_short_16","alias_value":"OHD3BRVLE4QAONNT","created_at":"2026-07-05T09:41:32.620617+00:00"},{"alias_kind":"pith_short_8","alias_value":"OHD3BRVL","created_at":"2026-07-05T09:41:32.620617+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.06033","citing_title":"Analysis of Blood Report Images Using General Purpose Vision-Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02271","citing_title":"Medical Report Generation: A Hierarchical Task Structure-Based Cross-Modal Causal Intervention Framework","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG","json":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG.json","graph_json":"https://pith.science/api/pith-number/OHD3BRVLE4QAONNTHEIB7HSNPG/graph.json","events_json":"https://pith.science/api/pith-number/OHD3BRVLE4QAONNTHEIB7HSNPG/events.json","paper":"https://pith.science/paper/OHD3BRVL"},"agent_actions":{"view_html":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG","download_json":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG.json","view_paper":"https://pith.science/paper/OHD3BRVL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.02469&json=true","fetch_graph":"https://pith.science/api/pith-number/OHD3BRVLE4QAONNTHEIB7HSNPG/graph.json","fetch_events":"https://pith.science/api/pith-number/OHD3BRVLE4QAONNTHEIB7HSNPG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG/action/storage_attestation","attest_author":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG/action/author_attestation","sign_citation":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG/action/citation_signature","submit_replication":"https://pith.science/pith/OHD3BRVLE4QAONNTHEIB7HSNPG/action/replication_record"}},"created_at":"2026-07-05T09:41:32.620617+00:00","updated_at":"2026-07-05T09:41:32.620617+00:00"}