{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NZJQEFHLUS7UBAMBDTTDT3ZQMI","short_pith_number":"pith:NZJQEFHL","schema_version":"1.0","canonical_sha256":"6e530214eba4bf4081811ce639ef306229fe29ea73e56f12938e85518d58efdc","source":{"kind":"arxiv","id":"2504.11695","version":4},"attestation_state":"computed","paper":{"title":"Interpreting the linear structure of vision-language model embedding spaces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Huangyuan Su, Isabel Papadimitriou, Sham Kakade, Stephanie Gil, Thomas Fel","submitted_at":"2025-04-16T01:40:06Z","abstract_excerpt":"Vision-language models encode images and text in a joint space, minimizing the distance between corresponding image and text pairs. How are language and images organized in this joint space, and how do the models encode meaning and modality? To investigate this, we train and release sparse autoencoders (SAEs) on the embedding spaces of four vision-language models (CLIP, SigLIP, SigLIP2, and AIMv2). SAEs approximate model embeddings as sparse linear combinations of learned directions, or \"concepts\". We find that, compared to other methods of linear feature learning, SAEs are better at reconstru"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.11695","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-16T01:40:06Z","cross_cats_sorted":["cs.CL","cs.MM"],"title_canon_sha256":"8a04e2c5143a3e41937a9581938fa157b0378caf548c2cd154ca230ac1b3cde0","abstract_canon_sha256":"08f9c2e5012230014364c016da86ab6b7ef12a64ae5b73f1588c945a8a8d668d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:57:35.920985Z","signature_b64":"/fzWufKf+xMiAcanGhIutG5azuXp82gf41a9DOzijyB3lBEPuf+LslYLNYc3xFZRe/DBpKlPzo9+KNo860ZMCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6e530214eba4bf4081811ce639ef306229fe29ea73e56f12938e85518d58efdc","last_reissued_at":"2026-07-05T11:57:35.920526Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:57:35.920526Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpreting the linear structure of vision-language model embedding spaces","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.MM"],"primary_cat":"cs.CV","authors_text":"Huangyuan Su, Isabel Papadimitriou, Sham Kakade, Stephanie Gil, Thomas Fel","submitted_at":"2025-04-16T01:40:06Z","abstract_excerpt":"Vision-language models encode images and text in a joint space, minimizing the distance between corresponding image and text pairs. How are language and images organized in this joint space, and how do the models encode meaning and modality? To investigate this, we train and release sparse autoencoders (SAEs) on the embedding spaces of four vision-language models (CLIP, SigLIP, SigLIP2, and AIMv2). SAEs approximate model embeddings as sparse linear combinations of learned directions, or \"concepts\". We find that, compared to other methods of linear feature learning, SAEs are better at reconstru"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.11695","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.11695/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.11695","created_at":"2026-07-05T11:57:35.920581+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.11695v4","created_at":"2026-07-05T11:57:35.920581+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.11695","created_at":"2026-07-05T11:57:35.920581+00:00"},{"alias_kind":"pith_short_12","alias_value":"NZJQEFHLUS7U","created_at":"2026-07-05T11:57:35.920581+00:00"},{"alias_kind":"pith_short_16","alias_value":"NZJQEFHLUS7UBAMB","created_at":"2026-07-05T11:57:35.920581+00:00"},{"alias_kind":"pith_short_8","alias_value":"NZJQEFHL","created_at":"2026-07-05T11:57:35.920581+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30815","citing_title":"When transformers learn \"impossible\" languages, what do they learn?","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11107","citing_title":"Birds of a Feather Flock Together: Background-Invariant Representations via Linear Structure in VLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06640","citing_title":"Concept-Based Abductive and Contrastive Explanations for Behaviors of Vision Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14363","citing_title":"The Cost of Language: Centroid Erasure Exposes and Exploits Modal Competition in Multimodal Language Models","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI","json":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI.json","graph_json":"https://pith.science/api/pith-number/NZJQEFHLUS7UBAMBDTTDT3ZQMI/graph.json","events_json":"https://pith.science/api/pith-number/NZJQEFHLUS7UBAMBDTTDT3ZQMI/events.json","paper":"https://pith.science/paper/NZJQEFHL"},"agent_actions":{"view_html":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI","download_json":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI.json","view_paper":"https://pith.science/paper/NZJQEFHL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.11695&json=true","fetch_graph":"https://pith.science/api/pith-number/NZJQEFHLUS7UBAMBDTTDT3ZQMI/graph.json","fetch_events":"https://pith.science/api/pith-number/NZJQEFHLUS7UBAMBDTTDT3ZQMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI/action/storage_attestation","attest_author":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI/action/author_attestation","sign_citation":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI/action/citation_signature","submit_replication":"https://pith.science/pith/NZJQEFHLUS7UBAMBDTTDT3ZQMI/action/replication_record"}},"created_at":"2026-07-05T11:57:35.920581+00:00","updated_at":"2026-07-05T11:57:35.920581+00:00"}