{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VFJPQNUGUQ6AIW3SYAH4DGSFTG","short_pith_number":"pith:VFJPQNUG","schema_version":"1.0","canonical_sha256":"a952f83686a43c045b72c00fc19a459983ae8096557f004d29bf462e0c672979","source":{"kind":"arxiv","id":"2403.16442","version":2},"attestation_state":"computed","paper":{"title":"If CLIP Could Talk: Understanding Vision-Language Model Representations Through Their Preferred Concept Descriptions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cristina Menghini, Reza Esfandiarpoor, Stephen H. Bach","submitted_at":"2024-03-25T06:05:50Z","abstract_excerpt":"Recent works often assume that Vision-Language Model (VLM) representations are based on visual attributes like shape. However, it is unclear to what extent VLMs prioritize this information to represent concepts. We propose Extract and Explore (EX2), a novel approach to characterize textual features that are important for VLMs. EX2 uses reinforcement learning to align a large language model with VLM preferences and generates descriptions that incorporate features that are important for the VLM. Then, we inspect the descriptions to identify features that contribute to VLM representations. Using "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.16442","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-25T06:05:50Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"0885bb5abf97bbf7e8aabd5bf24cfea1a65ee3124f333a3a2e5fdeab72f40779","abstract_canon_sha256":"5703d6504010cdf7008d9b94f44fa5bcaa1ccf2fed0b12d3ef32a2fd1464d55d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:33.816638Z","signature_b64":"UUjb8+7KMz7lpYjRmq5vwxOVbj6cfrLIB9x+IuMYGMeQIWccgLyraCAMAZp6QFVNE3G8r8I2dRSouBEsH6mjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a952f83686a43c045b72c00fc19a459983ae8096557f004d29bf462e0c672979","last_reissued_at":"2026-07-05T09:44:33.816095Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:33.816095Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"If CLIP Could Talk: Understanding Vision-Language Model Representations Through Their Preferred Concept Descriptions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cristina Menghini, Reza Esfandiarpoor, Stephen H. Bach","submitted_at":"2024-03-25T06:05:50Z","abstract_excerpt":"Recent works often assume that Vision-Language Model (VLM) representations are based on visual attributes like shape. However, it is unclear to what extent VLMs prioritize this information to represent concepts. We propose Extract and Explore (EX2), a novel approach to characterize textual features that are important for VLMs. EX2 uses reinforcement learning to align a large language model with VLM preferences and generates descriptions that incorporate features that are important for the VLM. Then, we inspect the descriptions to identify features that contribute to VLM representations. Using "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.16442","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.16442/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.16442","created_at":"2026-07-05T09:44:33.816154+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.16442v2","created_at":"2026-07-05T09:44:33.816154+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.16442","created_at":"2026-07-05T09:44:33.816154+00:00"},{"alias_kind":"pith_short_12","alias_value":"VFJPQNUGUQ6A","created_at":"2026-07-05T09:44:33.816154+00:00"},{"alias_kind":"pith_short_16","alias_value":"VFJPQNUGUQ6AIW3S","created_at":"2026-07-05T09:44:33.816154+00:00"},{"alias_kind":"pith_short_8","alias_value":"VFJPQNUG","created_at":"2026-07-05T09:44:33.816154+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.19757","citing_title":"Dual Risk Minimization: Towards Next-Level Robustness in Fine-tuning Zero-Shot Models","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG","json":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG.json","graph_json":"https://pith.science/api/pith-number/VFJPQNUGUQ6AIW3SYAH4DGSFTG/graph.json","events_json":"https://pith.science/api/pith-number/VFJPQNUGUQ6AIW3SYAH4DGSFTG/events.json","paper":"https://pith.science/paper/VFJPQNUG"},"agent_actions":{"view_html":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG","download_json":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG.json","view_paper":"https://pith.science/paper/VFJPQNUG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.16442&json=true","fetch_graph":"https://pith.science/api/pith-number/VFJPQNUGUQ6AIW3SYAH4DGSFTG/graph.json","fetch_events":"https://pith.science/api/pith-number/VFJPQNUGUQ6AIW3SYAH4DGSFTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG/action/storage_attestation","attest_author":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG/action/author_attestation","sign_citation":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG/action/citation_signature","submit_replication":"https://pith.science/pith/VFJPQNUGUQ6AIW3SYAH4DGSFTG/action/replication_record"}},"created_at":"2026-07-05T09:44:33.816154+00:00","updated_at":"2026-07-05T09:44:33.816154+00:00"}