{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:T6DEP5D5RZBQSEQC4YNBHXBPEQ","short_pith_number":"pith:T6DEP5D5","schema_version":"1.0","canonical_sha256":"9f8647f47d8e43091202e61a13dc2f243ee625392f46a38c888e3bb456ac8674","source":{"kind":"arxiv","id":"2209.15517","version":2},"attestation_state":"computed","paper":{"title":"Medical Image Understanding with Pretrained Vision Language Models: A Comprehensive Study","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huahui Yi, Kang Li, Qicheng Lao, Ziyuan Qin","submitted_at":"2022-09-30T15:06:13Z","abstract_excerpt":"The large-scale pre-trained vision language models (VLM) have shown remarkable domain transfer capability on natural images. However, it remains unknown whether this capability can also apply to the medical image domain. This paper thoroughly studies the knowledge transferability of pre-trained VLMs to the medical domain, where we show that well-designed medical prompts are the key to elicit knowledge from pre-trained VLMs. We demonstrate that by prompting with expressive attributes that are shared between domains, the VLM can carry the knowledge across domains and improve its generalization. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.15517","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-09-30T15:06:13Z","cross_cats_sorted":[],"title_canon_sha256":"834701fca2f6a7ae5e16783fafa0585654ea3a0760490a720f540815732c595b","abstract_canon_sha256":"08f7abcefbcbaf091e439e28114fc6a94b6a2ecd5eba667bb94ae67a8d84aa52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:39:21.109054Z","signature_b64":"YftCxBVCouzFKa4AKf6tmeyKE9ZC3p4zhteIigainQddc5rrKzbcFatdhCGosHeJzam2VKk1YJ/dAX258xFtCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f8647f47d8e43091202e61a13dc2f243ee625392f46a38c888e3bb456ac8674","last_reissued_at":"2026-07-05T05:39:21.108578Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:39:21.108578Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Medical Image Understanding with Pretrained Vision Language Models: A Comprehensive Study","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huahui Yi, Kang Li, Qicheng Lao, Ziyuan Qin","submitted_at":"2022-09-30T15:06:13Z","abstract_excerpt":"The large-scale pre-trained vision language models (VLM) have shown remarkable domain transfer capability on natural images. However, it remains unknown whether this capability can also apply to the medical image domain. This paper thoroughly studies the knowledge transferability of pre-trained VLMs to the medical domain, where we show that well-designed medical prompts are the key to elicit knowledge from pre-trained VLMs. We demonstrate that by prompting with expressive attributes that are shared between domains, the VLM can carry the knowledge across domains and improve its generalization. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.15517","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.15517/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.15517","created_at":"2026-07-05T05:39:21.108638+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.15517v2","created_at":"2026-07-05T05:39:21.108638+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.15517","created_at":"2026-07-05T05:39:21.108638+00:00"},{"alias_kind":"pith_short_12","alias_value":"T6DEP5D5RZBQ","created_at":"2026-07-05T05:39:21.108638+00:00"},{"alias_kind":"pith_short_16","alias_value":"T6DEP5D5RZBQSEQC","created_at":"2026-07-05T05:39:21.108638+00:00"},{"alias_kind":"pith_short_8","alias_value":"T6DEP5D5","created_at":"2026-07-05T05:39:21.108638+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.11411","citing_title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","ref_index":102,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ","json":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ.json","graph_json":"https://pith.science/api/pith-number/T6DEP5D5RZBQSEQC4YNBHXBPEQ/graph.json","events_json":"https://pith.science/api/pith-number/T6DEP5D5RZBQSEQC4YNBHXBPEQ/events.json","paper":"https://pith.science/paper/T6DEP5D5"},"agent_actions":{"view_html":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ","download_json":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ.json","view_paper":"https://pith.science/paper/T6DEP5D5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.15517&json=true","fetch_graph":"https://pith.science/api/pith-number/T6DEP5D5RZBQSEQC4YNBHXBPEQ/graph.json","fetch_events":"https://pith.science/api/pith-number/T6DEP5D5RZBQSEQC4YNBHXBPEQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ/action/storage_attestation","attest_author":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ/action/author_attestation","sign_citation":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ/action/citation_signature","submit_replication":"https://pith.science/pith/T6DEP5D5RZBQSEQC4YNBHXBPEQ/action/replication_record"}},"created_at":"2026-07-05T05:39:21.108638+00:00","updated_at":"2026-07-05T05:39:21.108638+00:00"}