{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K2ZTHHAQPYWPWIGGJHFMZOZ2XL","short_pith_number":"pith:K2ZTHHAQ","schema_version":"1.0","canonical_sha256":"56b3339c107e2cfb20c649caccbb3abade6a061efceeed437847c6c578b16c41","source":{"kind":"arxiv","id":"2304.05653","version":2},"attestation_state":"computed","paper":{"title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hualiang Wang, Jiheng Zhang, Xiaomeng Li, Yi Li, Yiqun Duan","submitted_at":"2023-04-12T07:16:55Z","abstract_excerpt":"Contrastive language-image pre-training (CLIP) is a powerful vision-language model that has shown great benefits for various tasks. However, we have identified some issues with its explainability, which undermine its credibility and limit the capacity for related tasks. Specifically, we find that CLIP tends to focus on background regions rather than foregrounds, with noisy activations at irrelevant positions on the visualization results. These phenomena conflict with conventional explainability methods based on the class attention map (CAM), where the raw model can highlight the local foregrou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.05653","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-04-12T07:16:55Z","cross_cats_sorted":[],"title_canon_sha256":"3f245b5d84555b1ec89fbb761574825035a8df63d30e36aa24122ac7e805a070","abstract_canon_sha256":"09c737ebd2aa5d9ffe34e354c75930bbe0e24f66cf440fbdeed92f84069562c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:23.893919Z","signature_b64":"6dnnDk2QMV9Bt3XQtxqi1P1T9NSKB0GZwKB+NJ7jhHTzf5XLf3aIRNVbrKMDNE6pDvvni00U6Wgjcv4/An6kBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56b3339c107e2cfb20c649caccbb3abade6a061efceeed437847c6c578b16c41","last_reissued_at":"2026-07-05T09:07:23.893465Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:23.893465Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hualiang Wang, Jiheng Zhang, Xiaomeng Li, Yi Li, Yiqun Duan","submitted_at":"2023-04-12T07:16:55Z","abstract_excerpt":"Contrastive language-image pre-training (CLIP) is a powerful vision-language model that has shown great benefits for various tasks. However, we have identified some issues with its explainability, which undermine its credibility and limit the capacity for related tasks. Specifically, we find that CLIP tends to focus on background regions rather than foregrounds, with noisy activations at irrelevant positions on the visualization results. These phenomena conflict with conventional explainability methods based on the class attention map (CAM), where the raw model can highlight the local foregrou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.05653","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.05653/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.05653","created_at":"2026-07-05T09:07:23.893529+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.05653v2","created_at":"2026-07-05T09:07:23.893529+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.05653","created_at":"2026-07-05T09:07:23.893529+00:00"},{"alias_kind":"pith_short_12","alias_value":"K2ZTHHAQPYWP","created_at":"2026-07-05T09:07:23.893529+00:00"},{"alias_kind":"pith_short_16","alias_value":"K2ZTHHAQPYWPWIGG","created_at":"2026-07-05T09:07:23.893529+00:00"},{"alias_kind":"pith_short_8","alias_value":"K2ZTHHAQ","created_at":"2026-07-05T09:07:23.893529+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.06818","citing_title":"Rethinking the Global Knowledge of CLIP in Training-Free Open-Vocabulary Semantic Segmentation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18816","citing_title":"Grad-ECLIP: Gradient-based Visual and Textual Explanations for CLIP","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09598","citing_title":"SoccerLens: Grounded Soccer Video Understanding Beyond Accuracy","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09598","citing_title":"SoccerLens: Grounded Soccer Video Understanding Beyond Accuracy","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05296","citing_title":"From Measurement to Mitigation: Quantifying and Reducing Identity Leakage in Image Representation Encoders with Linear Subspace Removal","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL","json":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL.json","graph_json":"https://pith.science/api/pith-number/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/graph.json","events_json":"https://pith.science/api/pith-number/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/events.json","paper":"https://pith.science/paper/K2ZTHHAQ"},"agent_actions":{"view_html":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL","download_json":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL.json","view_paper":"https://pith.science/paper/K2ZTHHAQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.05653&json=true","fetch_graph":"https://pith.science/api/pith-number/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/graph.json","fetch_events":"https://pith.science/api/pith-number/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/action/storage_attestation","attest_author":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/action/author_attestation","sign_citation":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/action/citation_signature","submit_replication":"https://pith.science/pith/K2ZTHHAQPYWPWIGGJHFMZOZ2XL/action/replication_record"}},"created_at":"2026-07-05T09:07:23.893529+00:00","updated_at":"2026-07-05T09:07:23.893529+00:00"}