{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KJ43RWG2PO2NMQQELIVI2PTAPC","short_pith_number":"pith:KJ43RWG2","schema_version":"1.0","canonical_sha256":"5279b8d8da7bb4d642045a2a8d3e60789fa875806a7ee646b09ebbc9a069ab16","source":{"kind":"arxiv","id":"2406.17639","version":3},"attestation_state":"computed","paper":{"title":"Mitigate the Gap: Investigating Approaches for Improving Cross-Modal Alignment in CLIP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Gerard de Melo, Sedigheh Eslami","submitted_at":"2024-06-25T15:24:02Z","abstract_excerpt":"Contrastive Language--Image Pre-training (CLIP) has manifested remarkable improvements in zero-shot classification and cross-modal vision-language tasks. Yet, from a geometrical point of view, the CLIP embedding space has been found to have a pronounced modality gap. This gap renders the embedding space overly sparse and disconnected, with different modalities being densely distributed in distinct subregions of the hypersphere. In this work, we aim at answering three main questions: 1. Does sharing the parameter space between the multi-modal encoders reduce the modality gap? 2. Can the gap be "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17639","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-25T15:24:02Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"8a21903b7f6130aac32739d5cba97a309ba7b9ff5dbe432906381323b7b2d242","abstract_canon_sha256":"a43f93c1af3ec3fad443e5c2177c21034dd01ef7903dceffc845c282b7b4b829"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:29.798588Z","signature_b64":"i5srSKIlQZAI2MBcreTuiUk/cak8k+/b0kSitksjOQf3LJfTbuNN5EI4YSMq9akmetJopq2iaOnJXryLaQiXAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5279b8d8da7bb4d642045a2a8d3e60789fa875806a7ee646b09ebbc9a069ab16","last_reissued_at":"2026-07-05T09:07:29.798093Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:29.798093Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigate the Gap: Investigating Approaches for Improving Cross-Modal Alignment in CLIP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Gerard de Melo, Sedigheh Eslami","submitted_at":"2024-06-25T15:24:02Z","abstract_excerpt":"Contrastive Language--Image Pre-training (CLIP) has manifested remarkable improvements in zero-shot classification and cross-modal vision-language tasks. Yet, from a geometrical point of view, the CLIP embedding space has been found to have a pronounced modality gap. This gap renders the embedding space overly sparse and disconnected, with different modalities being densely distributed in distinct subregions of the hypersphere. In this work, we aim at answering three main questions: 1. Does sharing the parameter space between the multi-modal encoders reduce the modality gap? 2. Can the gap be "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17639","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17639/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17639","created_at":"2026-07-05T09:07:29.798150+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17639v3","created_at":"2026-07-05T09:07:29.798150+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17639","created_at":"2026-07-05T09:07:29.798150+00:00"},{"alias_kind":"pith_short_12","alias_value":"KJ43RWG2PO2N","created_at":"2026-07-05T09:07:29.798150+00:00"},{"alias_kind":"pith_short_16","alias_value":"KJ43RWG2PO2NMQQE","created_at":"2026-07-05T09:07:29.798150+00:00"},{"alias_kind":"pith_short_8","alias_value":"KJ43RWG2","created_at":"2026-07-05T09:07:29.798150+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23137","citing_title":"STAMBRIDGE: Spectral-Temporal Amplitude-aware Mid-Feature Bridge for EEG Visual Decoding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23137","citing_title":"STAMBRIDGE: Spectral-Temporal Amplitude-aware Mid-Feature Bridge for EEG Visual Decoding","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.25383","citing_title":"CLIP-RD: Relative Distillation for Efficient CLIP Knowledge Distillation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11659","citing_title":"Reviving In-domain Fine-tuning Methods for Source-Free Cross-domain Few-shot Learning","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC","json":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC.json","graph_json":"https://pith.science/api/pith-number/KJ43RWG2PO2NMQQELIVI2PTAPC/graph.json","events_json":"https://pith.science/api/pith-number/KJ43RWG2PO2NMQQELIVI2PTAPC/events.json","paper":"https://pith.science/paper/KJ43RWG2"},"agent_actions":{"view_html":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC","download_json":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC.json","view_paper":"https://pith.science/paper/KJ43RWG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17639&json=true","fetch_graph":"https://pith.science/api/pith-number/KJ43RWG2PO2NMQQELIVI2PTAPC/graph.json","fetch_events":"https://pith.science/api/pith-number/KJ43RWG2PO2NMQQELIVI2PTAPC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC/action/storage_attestation","attest_author":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC/action/author_attestation","sign_citation":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC/action/citation_signature","submit_replication":"https://pith.science/pith/KJ43RWG2PO2NMQQELIVI2PTAPC/action/replication_record"}},"created_at":"2026-07-05T09:07:29.798150+00:00","updated_at":"2026-07-05T09:07:29.798150+00:00"}