{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UU4PJXBUUEJENSAF6MZ5NDDFQ4","short_pith_number":"pith:UU4PJXBU","schema_version":"1.0","canonical_sha256":"a538f4dc34a11246c805f333d68c65871cc5db9591d2a90ef8da67e54de62621","source":{"kind":"arxiv","id":"2407.19546","version":4},"attestation_state":"computed","paper":{"title":"MMCLIP: Cross-modal Attention Masked Modelling for Medical Language-Image Pre-Training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biao Wu, Ling Chen, Minh Hieu Phan, Qi Chen, Qi Wu, Yutong Xie, Zeyu Zhang","submitted_at":"2024-07-28T17:38:21Z","abstract_excerpt":"Vision-and-language pretraining (VLP) in the medical field utilizes contrastive learning on image-text pairs to achieve effective transfer across tasks. Yet, current VLP approaches with the masked modeling strategy face two challenges when applied to the medical domain. First, current models struggle to accurately reconstruct key pathological features due to the scarcity of medical data. Second, most methods only adopt either paired image-text or image-only data, failing to exploit the combination of both paired and unpaired data. To this end, this paper proposes the MMCLIP (Masked Medical Con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19546","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-28T17:38:21Z","cross_cats_sorted":[],"title_canon_sha256":"2b8a676dc15ac3cf4e65ac470f4a2e819a720928143afbbd872d3a13c5463513","abstract_canon_sha256":"a7dab31cc1a4fea964ad5cddd14e03e5d1aae799280ab58dc34ec5a7371aae77"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:45.772885Z","signature_b64":"ccqfDbcCJmQGU5lTAQhqli5gy7C6ZeD/J4QaZMjk1VLB5TSDhlN6jfDCT7UNb1jYrqUadnoLoOvu/A7ZjDXvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a538f4dc34a11246c805f333d68c65871cc5db9591d2a90ef8da67e54de62621","last_reissued_at":"2026-07-05T10:49:45.772367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:45.772367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMCLIP: Cross-modal Attention Masked Modelling for Medical Language-Image Pre-Training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biao Wu, Ling Chen, Minh Hieu Phan, Qi Chen, Qi Wu, Yutong Xie, Zeyu Zhang","submitted_at":"2024-07-28T17:38:21Z","abstract_excerpt":"Vision-and-language pretraining (VLP) in the medical field utilizes contrastive learning on image-text pairs to achieve effective transfer across tasks. Yet, current VLP approaches with the masked modeling strategy face two challenges when applied to the medical domain. First, current models struggle to accurately reconstruct key pathological features due to the scarcity of medical data. Second, most methods only adopt either paired image-text or image-only data, failing to exploit the combination of both paired and unpaired data. To this end, this paper proposes the MMCLIP (Masked Medical Con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19546","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19546/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19546","created_at":"2026-07-05T10:49:45.772427+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19546v4","created_at":"2026-07-05T10:49:45.772427+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19546","created_at":"2026-07-05T10:49:45.772427+00:00"},{"alias_kind":"pith_short_12","alias_value":"UU4PJXBUUEJE","created_at":"2026-07-05T10:49:45.772427+00:00"},{"alias_kind":"pith_short_16","alias_value":"UU4PJXBUUEJENSAF","created_at":"2026-07-05T10:49:45.772427+00:00"},{"alias_kind":"pith_short_8","alias_value":"UU4PJXBU","created_at":"2026-07-05T10:49:45.772427+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.09442","citing_title":"UIPress: Bringing Optical Token Compression to UI-to-Code Generation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17451","citing_title":"SegTTA: Training-Free Test-Time Augmentation for Zero-Shot Medical Imaging Segmentation","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4","json":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4.json","graph_json":"https://pith.science/api/pith-number/UU4PJXBUUEJENSAF6MZ5NDDFQ4/graph.json","events_json":"https://pith.science/api/pith-number/UU4PJXBUUEJENSAF6MZ5NDDFQ4/events.json","paper":"https://pith.science/paper/UU4PJXBU"},"agent_actions":{"view_html":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4","download_json":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4.json","view_paper":"https://pith.science/paper/UU4PJXBU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19546&json=true","fetch_graph":"https://pith.science/api/pith-number/UU4PJXBUUEJENSAF6MZ5NDDFQ4/graph.json","fetch_events":"https://pith.science/api/pith-number/UU4PJXBUUEJENSAF6MZ5NDDFQ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4/action/storage_attestation","attest_author":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4/action/author_attestation","sign_citation":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4/action/citation_signature","submit_replication":"https://pith.science/pith/UU4PJXBUUEJENSAF6MZ5NDDFQ4/action/replication_record"}},"created_at":"2026-07-05T10:49:45.772427+00:00","updated_at":"2026-07-05T10:49:45.772427+00:00"}