{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:OCK6PP3DUSQOHMPSCTQ6RN4OPX","short_pith_number":"pith:OCK6PP3D","schema_version":"1.0","canonical_sha256":"7095e7bf63a4a0e3b1f214e1e8b78e7dc667cf61ac67d2e656169a5bc89e0e84","source":{"kind":"arxiv","id":"2210.10163","version":1},"attestation_state":"computed","paper":{"title":"MedCLIP: Contrastive Learning from Unpaired Medical Images and Text","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dinesh Agarwal, Jimeng Sun, Zhenbang Wu, Zifeng Wang","submitted_at":"2022-10-18T21:06:29Z","abstract_excerpt":"Existing vision-text contrastive learning like CLIP aims to match the paired image and caption embeddings while pushing others apart, which improves representation transferability and supports zero-shot prediction. However, medical image-text datasets are orders of magnitude below the general images and captions from the internet. Moreover, previous methods encounter many false negatives, i.e., images and reports from separate patients probably carry the same semantics but are wrongly treated as negatives. In this paper, we decouple images and texts for multimodal contrastive learning thus sca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.10163","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-10-18T21:06:29Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"fede54f6a3a53192d3884d083ca01646256421924f308fb0520e39c3e2a7f460","abstract_canon_sha256":"9489fe10cacbcdbed2cb2b1ffb3c1351c6f11c618dbfb378e856c72bcd5a36ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:08:06.923799Z","signature_b64":"ip5UQyEMGKb3rUtdf//i4xqx8QLoyZ/SxQYprTg1Y7TYW6lecj6Yp9xbnifVRon9ZwY1prRWqKdmmZ1zIOAYCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7095e7bf63a4a0e3b1f214e1e8b78e7dc667cf61ac67d2e656169a5bc89e0e84","last_reissued_at":"2026-07-05T05:08:06.923382Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:08:06.923382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MedCLIP: Contrastive Learning from Unpaired Medical Images and Text","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dinesh Agarwal, Jimeng Sun, Zhenbang Wu, Zifeng Wang","submitted_at":"2022-10-18T21:06:29Z","abstract_excerpt":"Existing vision-text contrastive learning like CLIP aims to match the paired image and caption embeddings while pushing others apart, which improves representation transferability and supports zero-shot prediction. However, medical image-text datasets are orders of magnitude below the general images and captions from the internet. Moreover, previous methods encounter many false negatives, i.e., images and reports from separate patients probably carry the same semantics but are wrongly treated as negatives. In this paper, we decouple images and texts for multimodal contrastive learning thus sca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.10163","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.10163/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.10163","created_at":"2026-07-05T05:08:06.923446+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.10163v1","created_at":"2026-07-05T05:08:06.923446+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.10163","created_at":"2026-07-05T05:08:06.923446+00:00"},{"alias_kind":"pith_short_12","alias_value":"OCK6PP3DUSQO","created_at":"2026-07-05T05:08:06.923446+00:00"},{"alias_kind":"pith_short_16","alias_value":"OCK6PP3DUSQOHMPS","created_at":"2026-07-05T05:08:06.923446+00:00"},{"alias_kind":"pith_short_8","alias_value":"OCK6PP3D","created_at":"2026-07-05T05:08:06.923446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23487","citing_title":"CADRE: Stable, Parameter Efficient Adaptation of Medical Vision Language Models with Bounded Forgetting and Prior Drift","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29714","citing_title":"UniVAD v2: Unified Visual Anomaly Detection via Support-Conditioned Boundary Construction","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2411.05824","citing_title":"Navigating Distribution Shifts in Medical Image Analysis: A Survey","ref_index":203,"is_internal_anchor":false},{"citing_arxiv_id":"2503.01835","citing_title":"Primus: Enforcing Attention Usage for 3D Medical Image Segmentation","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14998","citing_title":"Tables Guide Vision: Learning to See the Heart through Tabular Data","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15736","citing_title":"BiomedAP: A Vision-Informed Dual-Anchor Framework with Gated Cross-Modal Fusion for Robust Medical Vision-Language Adaptation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19359","citing_title":"MAM-CLIP: Vision-Language Pretraining on Mammography Atlases for BI-RADS Classification","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2303.00915","citing_title":"BiomedCLIP: a multimodal biomedical foundation model pretrained from fifteen million scientific image-text pairs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11304","citing_title":"CheXTemporal: A Dataset for Temporally-Grounded Reasoning in Chest Radiography","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18713","citing_title":"Align then Refine: Text-Guided 3D Prostate Lesion Segmentation","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX","json":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX.json","graph_json":"https://pith.science/api/pith-number/OCK6PP3DUSQOHMPSCTQ6RN4OPX/graph.json","events_json":"https://pith.science/api/pith-number/OCK6PP3DUSQOHMPSCTQ6RN4OPX/events.json","paper":"https://pith.science/paper/OCK6PP3D"},"agent_actions":{"view_html":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX","download_json":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX.json","view_paper":"https://pith.science/paper/OCK6PP3D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.10163&json=true","fetch_graph":"https://pith.science/api/pith-number/OCK6PP3DUSQOHMPSCTQ6RN4OPX/graph.json","fetch_events":"https://pith.science/api/pith-number/OCK6PP3DUSQOHMPSCTQ6RN4OPX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX/action/storage_attestation","attest_author":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX/action/author_attestation","sign_citation":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX/action/citation_signature","submit_replication":"https://pith.science/pith/OCK6PP3DUSQOHMPSCTQ6RN4OPX/action/replication_record"}},"created_at":"2026-07-05T05:08:06.923446+00:00","updated_at":"2026-07-05T05:08:06.923446+00:00"}