{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZF243VS5V4CRJDGW75EAL43RCU","short_pith_number":"pith:ZF243VS5","schema_version":"1.0","canonical_sha256":"c975cdd65daf05148cd6ff4805f371151cc87aa537f2097fe6f358a830baae98","source":{"kind":"arxiv","id":"2310.05916","version":4},"attestation_state":"computed","paper":{"title":"Interpreting CLIP's Image Representation via Text-Based Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexei A. Efros, Jacob Steinhardt, Yossi Gandelsman","submitted_at":"2023-10-09T17:59:04Z","abstract_excerpt":"We investigate the CLIP image encoder by analyzing how individual model components affect the final representation. We decompose the image representation as a sum across individual image patches, model layers, and attention heads, and use CLIP's text representation to interpret the summands. Interpreting the attention heads, we characterize each head's role by automatically finding text representations that span its output space, which reveals property-specific roles for many heads (e.g. location or shape). Next, interpreting the image patches, we uncover an emergent spatial localization withi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.05916","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-09T17:59:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ab1913706ab914fe9faebcc6cee3d7c54bc3c15013f9ae40a7cbdd59e10389de","abstract_canon_sha256":"19aa51614a9d27616360d978a16618c4d9dd930b1bf560bc4721860ad204fda9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:02.235247Z","signature_b64":"ArHHhLTOEj0UA7msRZ1YSNrleGghGStQ+c3NQFE5YibjFZ5g+3kfU6NFDhU4cXY7ff8jJSw4FnVqIeKvaxkxCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c975cdd65daf05148cd6ff4805f371151cc87aa537f2097fe6f358a830baae98","last_reissued_at":"2026-07-05T08:02:02.234778Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:02.234778Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpreting CLIP's Image Representation via Text-Based Decomposition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexei A. Efros, Jacob Steinhardt, Yossi Gandelsman","submitted_at":"2023-10-09T17:59:04Z","abstract_excerpt":"We investigate the CLIP image encoder by analyzing how individual model components affect the final representation. We decompose the image representation as a sum across individual image patches, model layers, and attention heads, and use CLIP's text representation to interpret the summands. Interpreting the attention heads, we characterize each head's role by automatically finding text representations that span its output space, which reveals property-specific roles for many heads (e.g. location or shape). Next, interpreting the image patches, we uncover an emergent spatial localization withi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.05916","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.05916/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.05916","created_at":"2026-07-05T08:02:02.234833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.05916v4","created_at":"2026-07-05T08:02:02.234833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.05916","created_at":"2026-07-05T08:02:02.234833+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZF243VS5V4CR","created_at":"2026-07-05T08:02:02.234833+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZF243VS5V4CRJDGW","created_at":"2026-07-05T08:02:02.234833+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZF243VS5","created_at":"2026-07-05T08:02:02.234833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26460","citing_title":"AnchorDiff: Training-Free Concept Grounding for MM-DiTs via Anchor-Based Graph Propagation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20237","citing_title":"AnimeAdapter: A Modular Adapter for Appearance-Consistent Anime Character Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12485","citing_title":"Letting the neural code speak: Automated characterization of monkey visual neurons through human language","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03740","citing_title":"CLIP-SVD: Efficient and Interpretable Vision-Language Adaptation via Singular Values","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14837","citing_title":"V-SEAM: Visual Semantic Editing and Attention Modulating for Causal Interpretability of Vision-Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2403.19647","citing_title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12485","citing_title":"Letting the neural code speak: Automated characterization of monkey visual neurons through human language","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU","json":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU.json","graph_json":"https://pith.science/api/pith-number/ZF243VS5V4CRJDGW75EAL43RCU/graph.json","events_json":"https://pith.science/api/pith-number/ZF243VS5V4CRJDGW75EAL43RCU/events.json","paper":"https://pith.science/paper/ZF243VS5"},"agent_actions":{"view_html":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU","download_json":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU.json","view_paper":"https://pith.science/paper/ZF243VS5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.05916&json=true","fetch_graph":"https://pith.science/api/pith-number/ZF243VS5V4CRJDGW75EAL43RCU/graph.json","fetch_events":"https://pith.science/api/pith-number/ZF243VS5V4CRJDGW75EAL43RCU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU/action/storage_attestation","attest_author":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU/action/author_attestation","sign_citation":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU/action/citation_signature","submit_replication":"https://pith.science/pith/ZF243VS5V4CRJDGW75EAL43RCU/action/replication_record"}},"created_at":"2026-07-05T08:02:02.234833+00:00","updated_at":"2026-07-05T08:02:02.234833+00:00"}