{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:34FZWI6V32X6HELXQWBYFV7AAM","short_pith_number":"pith:34FZWI6V","schema_version":"1.0","canonical_sha256":"df0b9b23d5deafe39177858382d7e0033193a4b698ce11c0d477a9e5c8cec599","source":{"kind":"arxiv","id":"2404.09632","version":1},"attestation_state":"computed","paper":{"title":"Bridging Vision and Language Spaces with Assignment Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Jiyoung Lee, Jungin Park, Kwanghoon Sohn","submitted_at":"2024-04-15T10:04:15Z","abstract_excerpt":"This paper introduces VLAP, a novel approach that bridges pretrained vision models and large language models (LLMs) to make frozen LLMs understand the visual world. VLAP transforms the embedding space of pretrained vision models into the LLMs' word embedding space using a single linear layer for efficient and general-purpose visual and language understanding. Specifically, we harness well-established word embeddings to bridge two modality embedding spaces. The visual and text representations are simultaneously assigned to a set of word embeddings within pretrained LLMs by formulating the assig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.09632","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-15T10:04:15Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2cfeac53058366066a20d2988763b3d2f61d678df0a7ce71e7db24b2e4c49efc","abstract_canon_sha256":"8734ddc2449e04c35afddfac0fa5543828fe5dac27a209e462f9583740e19080"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:09.669127Z","signature_b64":"LMCIezruI4jdD+xbgDPg0R/+ydCbom/J/jM6WpB+4fuZcGaIPWlWKBVZMRK9D8N0SsTgM76oiMfsTu6b79GgBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df0b9b23d5deafe39177858382d7e0033193a4b698ce11c0d477a9e5c8cec599","last_reissued_at":"2026-07-05T08:08:09.668661Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:09.668661Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bridging Vision and Language Spaces with Assignment Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Jiyoung Lee, Jungin Park, Kwanghoon Sohn","submitted_at":"2024-04-15T10:04:15Z","abstract_excerpt":"This paper introduces VLAP, a novel approach that bridges pretrained vision models and large language models (LLMs) to make frozen LLMs understand the visual world. VLAP transforms the embedding space of pretrained vision models into the LLMs' word embedding space using a single linear layer for efficient and general-purpose visual and language understanding. Specifically, we harness well-established word embeddings to bridge two modality embedding spaces. The visual and text representations are simultaneously assigned to a set of word embeddings within pretrained LLMs by formulating the assig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.09632","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.09632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.09632","created_at":"2026-07-05T08:08:09.668718+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.09632v1","created_at":"2026-07-05T08:08:09.668718+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.09632","created_at":"2026-07-05T08:08:09.668718+00:00"},{"alias_kind":"pith_short_12","alias_value":"34FZWI6V32X6","created_at":"2026-07-05T08:08:09.668718+00:00"},{"alias_kind":"pith_short_16","alias_value":"34FZWI6V32X6HELX","created_at":"2026-07-05T08:08:09.668718+00:00"},{"alias_kind":"pith_short_8","alias_value":"34FZWI6V","created_at":"2026-07-05T08:08:09.668718+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.06895","citing_title":"BASIC: Boosting Visual Alignment with Intrinsic Refined Embeddings in Multimodal Large Language Models","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM","json":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM.json","graph_json":"https://pith.science/api/pith-number/34FZWI6V32X6HELXQWBYFV7AAM/graph.json","events_json":"https://pith.science/api/pith-number/34FZWI6V32X6HELXQWBYFV7AAM/events.json","paper":"https://pith.science/paper/34FZWI6V"},"agent_actions":{"view_html":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM","download_json":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM.json","view_paper":"https://pith.science/paper/34FZWI6V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.09632&json=true","fetch_graph":"https://pith.science/api/pith-number/34FZWI6V32X6HELXQWBYFV7AAM/graph.json","fetch_events":"https://pith.science/api/pith-number/34FZWI6V32X6HELXQWBYFV7AAM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM/action/storage_attestation","attest_author":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM/action/author_attestation","sign_citation":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM/action/citation_signature","submit_replication":"https://pith.science/pith/34FZWI6V32X6HELXQWBYFV7AAM/action/replication_record"}},"created_at":"2026-07-05T08:08:09.668718+00:00","updated_at":"2026-07-05T08:08:09.668718+00:00"}