{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5VHMRDIZA5GZXJQJIQMZIAQRV6","short_pith_number":"pith:5VHMRDIZ","schema_version":"1.0","canonical_sha256":"ed4ec88d19074d9ba6094419940211af803a63e405b63f4137d7eb4457029e63","source":{"kind":"arxiv","id":"2409.19684","version":1},"attestation_state":"computed","paper":{"title":"MedViLaM: A multimodal large language model with advanced generalizability and explainability for medical data understanding and generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hao Sun, Hongsheng Li, Lijian Xu, Shaoting Zhang, Ziyu Ni","submitted_at":"2024-09-29T12:23:10Z","abstract_excerpt":"Medicine is inherently multimodal and multitask, with diverse data modalities spanning text, imaging. However, most models in medical field are unimodal single tasks and lack good generalizability and explainability. In this study, we introduce MedViLaM, a unified vision-language model towards a generalist model for medical data that can flexibly encode and interpret various forms of medical data, including clinical language and imaging, all using the same set of model weights. To facilitate the creation of such multi-task model, we have curated MultiMedBench, a comprehensive pretaining datase"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19684","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-09-29T12:23:10Z","cross_cats_sorted":[],"title_canon_sha256":"498554972add9a4b91e18b259523e38f4a9e64fb75421f930384b7b453cb0732","abstract_canon_sha256":"ee18e426e916fb2d6f29d731f6d62e6fba89dbc488e0cfbfa72fceae1a9a1e65"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:20.904654Z","signature_b64":"rbqXiHcpXoLbO++rVKSva78gCuqf4Gohav263mKcy64JVd3sttPtXcNH5C1oj9aV577OJXP1PQ+2jReXSVR3AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed4ec88d19074d9ba6094419940211af803a63e405b63f4137d7eb4457029e63","last_reissued_at":"2026-07-05T09:13:20.903682Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:20.903682Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedViLaM: A multimodal large language model with advanced generalizability and explainability for medical data understanding and generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hao Sun, Hongsheng Li, Lijian Xu, Shaoting Zhang, Ziyu Ni","submitted_at":"2024-09-29T12:23:10Z","abstract_excerpt":"Medicine is inherently multimodal and multitask, with diverse data modalities spanning text, imaging. However, most models in medical field are unimodal single tasks and lack good generalizability and explainability. In this study, we introduce MedViLaM, a unified vision-language model towards a generalist model for medical data that can flexibly encode and interpret various forms of medical data, including clinical language and imaging, all using the same set of model weights. To facilitate the creation of such multi-task model, we have curated MultiMedBench, a comprehensive pretaining datase"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19684","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19684","created_at":"2026-07-05T09:13:20.903752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19684v1","created_at":"2026-07-05T09:13:20.903752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19684","created_at":"2026-07-05T09:13:20.903752+00:00"},{"alias_kind":"pith_short_12","alias_value":"5VHMRDIZA5GZ","created_at":"2026-07-05T09:13:20.903752+00:00"},{"alias_kind":"pith_short_16","alias_value":"5VHMRDIZA5GZXJQJ","created_at":"2026-07-05T09:13:20.903752+00:00"},{"alias_kind":"pith_short_8","alias_value":"5VHMRDIZ","created_at":"2026-07-05T09:13:20.903752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08641","citing_title":"Learnable Token Sparsification for Efficient Gigapixel Whole Slide Image Reasoning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03417","citing_title":"A unified multi-task framework enables interpretable chest radiograph analysis","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28051","citing_title":"Beyond Surrogate Gradients: Fully Differentiable Token Pruning for Vision-Language Models","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02695","citing_title":"XrayClaw: Cooperative-Competitive Multi-Agent Alignment for Trustworthy Chest X-ray Diagnosis","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6","json":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6.json","graph_json":"https://pith.science/api/pith-number/5VHMRDIZA5GZXJQJIQMZIAQRV6/graph.json","events_json":"https://pith.science/api/pith-number/5VHMRDIZA5GZXJQJIQMZIAQRV6/events.json","paper":"https://pith.science/paper/5VHMRDIZ"},"agent_actions":{"view_html":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6","download_json":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6.json","view_paper":"https://pith.science/paper/5VHMRDIZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19684&json=true","fetch_graph":"https://pith.science/api/pith-number/5VHMRDIZA5GZXJQJIQMZIAQRV6/graph.json","fetch_events":"https://pith.science/api/pith-number/5VHMRDIZA5GZXJQJIQMZIAQRV6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6/action/storage_attestation","attest_author":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6/action/author_attestation","sign_citation":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6/action/citation_signature","submit_replication":"https://pith.science/pith/5VHMRDIZA5GZXJQJIQMZIAQRV6/action/replication_record"}},"created_at":"2026-07-05T09:13:20.903752+00:00","updated_at":"2026-07-05T09:13:20.903752+00:00"}