{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MNDJGFLDQNF7TVCBJ2YK473SDP","short_pith_number":"pith:MNDJGFLD","schema_version":"1.0","canonical_sha256":"6346931563834bf9d4414eb0ae7f721bec71b467a98794c3121691121b358ec8","source":{"kind":"arxiv","id":"2506.11976","version":2},"attestation_state":"computed","paper":{"title":"How Visual Representations Map to Language Feature Space in Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Ashkan Khakzar, Constantin Venhoff, Neel Nanda, Philip Torr, Sonia Joseph","submitted_at":"2025-06-13T17:34:05Z","abstract_excerpt":"Effective multimodal reasoning depends on the alignment of visual and linguistic representations, yet the mechanisms by which vision-language models (VLMs) achieve this alignment remain poorly understood. Following the LiMBeR framework, we deliberately maintain a frozen large language model (LLM) and a frozen vision transformer (ViT), connected solely by training a linear adapter during visual instruction tuning. By keeping the language model frozen, we ensure it maintains its original language representations without adaptation to visual data. Consequently, the linear adapter must map visual "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.11976","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-13T17:34:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"861f5f0e4c1d1309bd00efca3b68b99792902a50138c20f0bb74566a276ea1dd","abstract_canon_sha256":"503c5caf82a23cadef5168e4e0014f63f10344188f09af4b08bcef0d7ac10ac7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:12.538982Z","signature_b64":"/lLj/fCGXMAERERMY6Cn2l8/noGLwgbnc1cqsqAyGNn32DIcFeY7jUso4ZJrLpfgeU5WN2hbU4YDA7EMC3ZtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6346931563834bf9d4414eb0ae7f721bec71b467a98794c3121691121b358ec8","last_reissued_at":"2026-07-05T11:25:12.538430Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:12.538430Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Visual Representations Map to Language Feature Space in Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Ashkan Khakzar, Constantin Venhoff, Neel Nanda, Philip Torr, Sonia Joseph","submitted_at":"2025-06-13T17:34:05Z","abstract_excerpt":"Effective multimodal reasoning depends on the alignment of visual and linguistic representations, yet the mechanisms by which vision-language models (VLMs) achieve this alignment remain poorly understood. Following the LiMBeR framework, we deliberately maintain a frozen large language model (LLM) and a frozen vision transformer (ViT), connected solely by training a linear adapter during visual instruction tuning. By keeping the language model frozen, we ensure it maintains its original language representations without adaptation to visual data. Consequently, the linear adapter must map visual "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.11976","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.11976/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.11976","created_at":"2026-07-05T11:25:12.538495+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.11976v2","created_at":"2026-07-05T11:25:12.538495+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.11976","created_at":"2026-07-05T11:25:12.538495+00:00"},{"alias_kind":"pith_short_12","alias_value":"MNDJGFLDQNF7","created_at":"2026-07-05T11:25:12.538495+00:00"},{"alias_kind":"pith_short_16","alias_value":"MNDJGFLDQNF7TVCB","created_at":"2026-07-05T11:25:12.538495+00:00"},{"alias_kind":"pith_short_8","alias_value":"MNDJGFLD","created_at":"2026-07-05T11:25:12.538495+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08511","citing_title":"Look Less, Reason More: Block-wise Attention Skipping for Efficient Multimodal LLMs","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP","json":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP.json","graph_json":"https://pith.science/api/pith-number/MNDJGFLDQNF7TVCBJ2YK473SDP/graph.json","events_json":"https://pith.science/api/pith-number/MNDJGFLDQNF7TVCBJ2YK473SDP/events.json","paper":"https://pith.science/paper/MNDJGFLD"},"agent_actions":{"view_html":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP","download_json":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP.json","view_paper":"https://pith.science/paper/MNDJGFLD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.11976&json=true","fetch_graph":"https://pith.science/api/pith-number/MNDJGFLDQNF7TVCBJ2YK473SDP/graph.json","fetch_events":"https://pith.science/api/pith-number/MNDJGFLDQNF7TVCBJ2YK473SDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP/action/storage_attestation","attest_author":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP/action/author_attestation","sign_citation":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP/action/citation_signature","submit_replication":"https://pith.science/pith/MNDJGFLDQNF7TVCBJ2YK473SDP/action/replication_record"}},"created_at":"2026-07-05T11:25:12.538495+00:00","updated_at":"2026-07-05T11:25:12.538495+00:00"}