{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Z4TQB6YC3TXLLAHA7QBZ3ZNN3X","short_pith_number":"pith:Z4TQB6YC","schema_version":"1.0","canonical_sha256":"cf2700fb02dceeb580e0fc039de5addde7f2b8700fe6f47664292a0710b86e19","source":{"kind":"arxiv","id":"2508.12108","version":1},"attestation_state":"computed","paper":{"title":"VELVET-Med: Vision and Efficient Language Pre-training for Volumetric Imaging Tasks in Medicine","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Si Yong Yeo, Xulei Yang, Yang Yu, Ziyang Zhang","submitted_at":"2025-08-16T17:08:43Z","abstract_excerpt":"Vision-and-language models (VLMs) have been increasingly explored in the medical domain, particularly following the success of CLIP in general domain. However, unlike the relatively straightforward pairing of 2D images and text, curating large-scale paired data in the medical field for volumetric modalities such as CT scans remains a challenging and time-intensive process. This difficulty often limits the performance on downstream tasks. To address these challenges, we propose a novel vision-language pre-training (VLP) framework, termed as \\textbf{VELVET-Med}, specifically designed for limited"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.12108","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-08-16T17:08:43Z","cross_cats_sorted":[],"title_canon_sha256":"540a196dd306beb64f19dad3518175a591f1b0992155fd108ebabb92b5da0ec2","abstract_canon_sha256":"8a7fbe3b7bc7ebc1fbef3d032ce2eeaf75132e2b3542cb4204b765d2e8f6f0c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:00.202220Z","signature_b64":"Fk6I+Mc5UNQnLJT7dE4d9RdnIGLjkVn7yhxbP/jVhSmOAOkcyVzHOk3eseE0vWe7RovpBZ8WXG2YAyehqRPRDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf2700fb02dceeb580e0fc039de5addde7f2b8700fe6f47664292a0710b86e19","last_reissued_at":"2026-07-05T11:55:00.201712Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:00.201712Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VELVET-Med: Vision and Efficient Language Pre-training for Volumetric Imaging Tasks in Medicine","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Si Yong Yeo, Xulei Yang, Yang Yu, Ziyang Zhang","submitted_at":"2025-08-16T17:08:43Z","abstract_excerpt":"Vision-and-language models (VLMs) have been increasingly explored in the medical domain, particularly following the success of CLIP in general domain. However, unlike the relatively straightforward pairing of 2D images and text, curating large-scale paired data in the medical field for volumetric modalities such as CT scans remains a challenging and time-intensive process. This difficulty often limits the performance on downstream tasks. To address these challenges, we propose a novel vision-language pre-training (VLP) framework, termed as \\textbf{VELVET-Med}, specifically designed for limited"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.12108","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.12108/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.12108","created_at":"2026-07-05T11:55:00.201791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.12108v1","created_at":"2026-07-05T11:55:00.201791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.12108","created_at":"2026-07-05T11:55:00.201791+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z4TQB6YC3TXL","created_at":"2026-07-05T11:55:00.201791+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z4TQB6YC3TXLLAHA","created_at":"2026-07-05T11:55:00.201791+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z4TQB6YC","created_at":"2026-07-05T11:55:00.201791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X","json":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X.json","graph_json":"https://pith.science/api/pith-number/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/graph.json","events_json":"https://pith.science/api/pith-number/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/events.json","paper":"https://pith.science/paper/Z4TQB6YC"},"agent_actions":{"view_html":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X","download_json":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X.json","view_paper":"https://pith.science/paper/Z4TQB6YC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.12108&json=true","fetch_graph":"https://pith.science/api/pith-number/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/graph.json","fetch_events":"https://pith.science/api/pith-number/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/action/storage_attestation","attest_author":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/action/author_attestation","sign_citation":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/action/citation_signature","submit_replication":"https://pith.science/pith/Z4TQB6YC3TXLLAHA7QBZ3ZNN3X/action/replication_record"}},"created_at":"2026-07-05T11:55:00.201791+00:00","updated_at":"2026-07-05T11:55:00.201791+00:00"}