{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7OX6JSW7UE5XVD5OXOXWJVSTMD","short_pith_number":"pith:7OX6JSW7","schema_version":"1.0","canonical_sha256":"fbafe4cadfa13b7a8faebbaf64d65360d94942a58fa8a7288f00102349da91f7","source":{"kind":"arxiv","id":"2411.12915","version":3},"attestation_state":"computed","paper":{"title":"VILA-M3: Enhancing Vision-Language Models with Medical Expert Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andriy Myronenko, Baris Turkbey, Benjamin Simon, Can Zhao, Daguang Xu, Dong Yang, Greg Heinrich, Holger Roth, Hongxu Yin, Marc Edgar, Michael Zephyr, Mingxin Zheng, Pavlo Molchanov, Pengfei Guo, Song Han, Stephanie Harmon, Stephen Aylward, Vishwesh Nath, Wenqi Li, Yao Lu, Yee Man Law, Yucheng Tang, Yufan He, Zhijian Liu, Ziyue Xu","submitted_at":"2024-11-19T22:59:14Z","abstract_excerpt":"Generalist vision language models (VLMs) have made significant strides in computer vision, but they fall short in specialized fields like healthcare, where expert knowledge is essential. In traditional computer vision tasks, creative or approximate answers may be acceptable, but in healthcare, precision is paramount.Current large multimodal models like Gemini and GPT-4o are insufficient for medical tasks due to their reliance on memorized internet knowledge rather than the nuanced expertise required in healthcare. VLMs are usually trained in three stages: vision pre-training, vision-language p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.12915","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-19T22:59:14Z","cross_cats_sorted":[],"title_canon_sha256":"c83018efca1b8bccc9cfea4225843ce4470c649ce25959edb7464fc0d2bbd637","abstract_canon_sha256":"456ae1f05cb7deb1df4012f7d06719d04429dccb761bf540bde81f4a202e8051"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:58.055267Z","signature_b64":"4fDm1hY7JQwh7xDnIvDAoIqd5bMUpvyqr9TnT8NsSwgDXm+IIxHXiUJ/sdNJdJuRdx3Gmg/3/PGIz2yQkpJaBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fbafe4cadfa13b7a8faebbaf64d65360d94942a58fa8a7288f00102349da91f7","last_reissued_at":"2026-07-05T10:23:58.054756Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:58.054756Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VILA-M3: Enhancing Vision-Language Models with Medical Expert Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andriy Myronenko, Baris Turkbey, Benjamin Simon, Can Zhao, Daguang Xu, Dong Yang, Greg Heinrich, Holger Roth, Hongxu Yin, Marc Edgar, Michael Zephyr, Mingxin Zheng, Pavlo Molchanov, Pengfei Guo, Song Han, Stephanie Harmon, Stephen Aylward, Vishwesh Nath, Wenqi Li, Yao Lu, Yee Man Law, Yucheng Tang, Yufan He, Zhijian Liu, Ziyue Xu","submitted_at":"2024-11-19T22:59:14Z","abstract_excerpt":"Generalist vision language models (VLMs) have made significant strides in computer vision, but they fall short in specialized fields like healthcare, where expert knowledge is essential. In traditional computer vision tasks, creative or approximate answers may be acceptable, but in healthcare, precision is paramount.Current large multimodal models like Gemini and GPT-4o are insufficient for medical tasks due to their reliance on memorized internet knowledge rather than the nuanced expertise required in healthcare. VLMs are usually trained in three stages: vision pre-training, vision-language p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.12915","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.12915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.12915","created_at":"2026-07-05T10:23:58.054813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.12915v3","created_at":"2026-07-05T10:23:58.054813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.12915","created_at":"2026-07-05T10:23:58.054813+00:00"},{"alias_kind":"pith_short_12","alias_value":"7OX6JSW7UE5X","created_at":"2026-07-05T10:23:58.054813+00:00"},{"alias_kind":"pith_short_16","alias_value":"7OX6JSW7UE5XVD5O","created_at":"2026-07-05T10:23:58.054813+00:00"},{"alias_kind":"pith_short_8","alias_value":"7OX6JSW7","created_at":"2026-07-05T10:23:58.054813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.04468","citing_title":"NVILA: Efficient Frontier Visual Language Models","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD","json":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD.json","graph_json":"https://pith.science/api/pith-number/7OX6JSW7UE5XVD5OXOXWJVSTMD/graph.json","events_json":"https://pith.science/api/pith-number/7OX6JSW7UE5XVD5OXOXWJVSTMD/events.json","paper":"https://pith.science/paper/7OX6JSW7"},"agent_actions":{"view_html":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD","download_json":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD.json","view_paper":"https://pith.science/paper/7OX6JSW7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.12915&json=true","fetch_graph":"https://pith.science/api/pith-number/7OX6JSW7UE5XVD5OXOXWJVSTMD/graph.json","fetch_events":"https://pith.science/api/pith-number/7OX6JSW7UE5XVD5OXOXWJVSTMD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD/action/storage_attestation","attest_author":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD/action/author_attestation","sign_citation":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD/action/citation_signature","submit_replication":"https://pith.science/pith/7OX6JSW7UE5XVD5OXOXWJVSTMD/action/replication_record"}},"created_at":"2026-07-05T10:23:58.054813+00:00","updated_at":"2026-07-05T10:23:58.054813+00:00"}