{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4SXETXTZ4FJJOUI6LWPK7W44FP","short_pith_number":"pith:4SXETXTZ","schema_version":"1.0","canonical_sha256":"e4ae49de79e15297511e5d9eafdb9c2be0aec5df4bb1a5ad896009976a68b7b6","source":{"kind":"arxiv","id":"2401.06209","version":2},"attestation_state":"computed","paper":{"title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Saining Xie, Shengbang Tong, Yann LeCun, Yi Ma, Yuexiang Zhai, Zhuang Liu","submitted_at":"2024-01-11T18:58:36Z","abstract_excerpt":"Is vision good enough for language? Recent advancements in multimodal models primarily stem from the powerful reasoning abilities of large language models (LLMs). However, the visual component typically depends only on the instance-level contrastive language-image pre-training (CLIP). Our research reveals that the visual capabilities in recent multimodal LLMs (MLLMs) still exhibit systematic shortcomings. To understand the roots of these errors, we explore the gap between the visual embedding space of CLIP and vision-only self-supervised learning. We identify ''CLIP-blind pairs'' - images that"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06209","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-11T18:58:36Z","cross_cats_sorted":[],"title_canon_sha256":"d7adf01358939a98456ed8509cc278be6b5b8484f4e73703458becff6e447011","abstract_canon_sha256":"f154228ff448cd08939a760558b2f8224efda56af4f77482a10fe93c44217268"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:11:56.770249Z","signature_b64":"KMfpzfT9HiKoSflHETUD1vHxZ898JnW4XYQU6Qqz56ux9OCBIwTbKm/MYJDEWvBAxkyZ+/kNhOcTFSRNCkpCBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4ae49de79e15297511e5d9eafdb9c2be0aec5df4bb1a5ad896009976a68b7b6","last_reissued_at":"2026-07-05T08:11:56.769812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:11:56.769812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Saining Xie, Shengbang Tong, Yann LeCun, Yi Ma, Yuexiang Zhai, Zhuang Liu","submitted_at":"2024-01-11T18:58:36Z","abstract_excerpt":"Is vision good enough for language? Recent advancements in multimodal models primarily stem from the powerful reasoning abilities of large language models (LLMs). However, the visual component typically depends only on the instance-level contrastive language-image pre-training (CLIP). Our research reveals that the visual capabilities in recent multimodal LLMs (MLLMs) still exhibit systematic shortcomings. To understand the roots of these errors, we explore the gap between the visual embedding space of CLIP and vision-only self-supervised learning. We identify ''CLIP-blind pairs'' - images that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06209","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06209/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06209","created_at":"2026-07-05T08:11:56.769867+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06209v2","created_at":"2026-07-05T08:11:56.769867+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06209","created_at":"2026-07-05T08:11:56.769867+00:00"},{"alias_kind":"pith_short_12","alias_value":"4SXETXTZ4FJJ","created_at":"2026-07-05T08:11:56.769867+00:00"},{"alias_kind":"pith_short_16","alias_value":"4SXETXTZ4FJJOUI6","created_at":"2026-07-05T08:11:56.769867+00:00"},{"alias_kind":"pith_short_8","alias_value":"4SXETXTZ","created_at":"2026-07-05T08:11:56.769867+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18974","citing_title":"Visual-OPSD: Cross-Modal On-Policy Self-Distillation for Efficient Unified Multimodal Reasoning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11854","citing_title":"Fine-tuning Multi-modal LLMs with ART: Art-based Reinforcement Training","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00384","citing_title":"VESTA: Visual Exploration with Statistical Tool Agents","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26380","citing_title":"VisualNeedle: Benchmarking Active Visual Search in Information-Dense Scenes","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.16009","citing_title":"Bridging the Usability Gap: Lessons from Interpreting Studies for Machine Interpreting Design","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20665","citing_title":"The Expense of Seeing: Attaining Trustworthy Multimodal Reasoning Within the Monolithic Paradigm","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01955","citing_title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21678","citing_title":"Agentic Learner with Grow-and-Refine Multimodal Semantic Memory","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13310","citing_title":"Visual Para-Thinker: Divide-and-Conquer Reasoning for Visual Comprehension","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2403.05525","citing_title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20665","citing_title":"The Expense of Seeing: Attaining Trustworthy Multimodal Reasoning Within the Monolithic Paradigm","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10219","citing_title":"Cognitive Pivot Points and Visual Anchoring: Unveiling and Rectifying Hallucinations in Multimodal Reasoning Models","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02276","citing_title":"Kimi K2.5: Visual Agentic Intelligence","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13803","citing_title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16883","citing_title":"SinkRouter: Sink-Aware Routing for Efficient Long-Context Decoding in Large Language and Multimodal Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":251,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP","json":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP.json","graph_json":"https://pith.science/api/pith-number/4SXETXTZ4FJJOUI6LWPK7W44FP/graph.json","events_json":"https://pith.science/api/pith-number/4SXETXTZ4FJJOUI6LWPK7W44FP/events.json","paper":"https://pith.science/paper/4SXETXTZ"},"agent_actions":{"view_html":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP","download_json":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP.json","view_paper":"https://pith.science/paper/4SXETXTZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06209&json=true","fetch_graph":"https://pith.science/api/pith-number/4SXETXTZ4FJJOUI6LWPK7W44FP/graph.json","fetch_events":"https://pith.science/api/pith-number/4SXETXTZ4FJJOUI6LWPK7W44FP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP/action/storage_attestation","attest_author":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP/action/author_attestation","sign_citation":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP/action/citation_signature","submit_replication":"https://pith.science/pith/4SXETXTZ4FJJOUI6LWPK7W44FP/action/replication_record"}},"created_at":"2026-07-05T08:11:56.769867+00:00","updated_at":"2026-07-05T08:11:56.769867+00:00"}