{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QZC3UH5JBJ3XKV2VBIHEMH7BXG","short_pith_number":"pith:QZC3UH5J","schema_version":"1.0","canonical_sha256":"8645ba1fa90a777557550a0e461fe1b9b3f09d902938f72b022d4aed903e1b80","source":{"kind":"arxiv","id":"2310.08825","version":3},"attestation_state":"computed","paper":{"title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongsheng Jiang, Hao Zhang, Hongkai Xiong, Jin'e Zhao, Jin Li, Songlin Liu, Xiaopeng Zhang, Yuchen Liu, Zhen Gao","submitted_at":"2023-10-13T02:41:55Z","abstract_excerpt":"Multi-modal Large Language Models (MLLMs) have made significant strides in expanding the capabilities of Large Language Models (LLMs) through the incorporation of visual perception interfaces. Despite the emergence of exciting applications and the availability of diverse instruction tuning data, existing approaches often rely on CLIP or its variants as the visual branch, and merely extract features from the deep layers. However, these methods lack a comprehensive analysis of the visual encoders in MLLMs. In this paper, we conduct an extensive investigation into the effectiveness of different v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.08825","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-13T02:41:55Z","cross_cats_sorted":[],"title_canon_sha256":"a1aed792fe6f650563d925e68bb2a2864444798a2bf569a276ea98dc51a279c3","abstract_canon_sha256":"17794eee8fb6d39abdf8b34c56d116aeca43905186cf7c3cec537c6f1e3bea2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:31.519583Z","signature_b64":"kZfGvQ/E962MkbHH+OdTWzckFM3UKlq13sI9My6XJb+kXG05gHc4x/eQ7jEZmZO9vq0HKUoiQ/KwDX169vs8Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8645ba1fa90a777557550a0e461fe1b9b3f09d902938f72b022d4aed903e1b80","last_reissued_at":"2026-07-05T07:53:31.519140Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:31.519140Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongsheng Jiang, Hao Zhang, Hongkai Xiong, Jin'e Zhao, Jin Li, Songlin Liu, Xiaopeng Zhang, Yuchen Liu, Zhen Gao","submitted_at":"2023-10-13T02:41:55Z","abstract_excerpt":"Multi-modal Large Language Models (MLLMs) have made significant strides in expanding the capabilities of Large Language Models (LLMs) through the incorporation of visual perception interfaces. Despite the emergence of exciting applications and the availability of diverse instruction tuning data, existing approaches often rely on CLIP or its variants as the visual branch, and merely extract features from the deep layers. However, these methods lack a comprehensive analysis of the visual encoders in MLLMs. In this paper, we conduct an extensive investigation into the effectiveness of different v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.08825","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.08825/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.08825","created_at":"2026-07-05T07:53:31.519200+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.08825v3","created_at":"2026-07-05T07:53:31.519200+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.08825","created_at":"2026-07-05T07:53:31.519200+00:00"},{"alias_kind":"pith_short_12","alias_value":"QZC3UH5JBJ3X","created_at":"2026-07-05T07:53:31.519200+00:00"},{"alias_kind":"pith_short_16","alias_value":"QZC3UH5JBJ3XKV2V","created_at":"2026-07-05T07:53:31.519200+00:00"},{"alias_kind":"pith_short_8","alias_value":"QZC3UH5J","created_at":"2026-07-05T07:53:31.519200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24467","citing_title":"CompressKV: Semantic-Retrieval-Guided KV-Cache Compression for Resource-Efficient Long-Context LLM Inference","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22237","citing_title":"Investigating The Security of Modern AI and Cloud Infrastructure","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23693","citing_title":"EXPO-SQL: Execution-based Clause-level Policy Optimization for Text-to-SQL","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10977","citing_title":"PASA: A Principled Embedding-Space Watermarking Approach for LLM-Generated Text under Semantic-Invariant Attacks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24675","citing_title":"VaaWIT: Visual-Aware Adaptation of Large Language Models for Multilingual Web Image Translation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11051","citing_title":"NP-LoRA: Null Space Projection for Subject-Style LoRA Fusion","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21210","citing_title":"Toward Generalizable Forgery Detection and Reasoning","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2504.07415","citing_title":"RA-RRG: Multimodal Retrieval-Augmented Radiology Report Generation with Key Phrase Extraction","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17128","citing_title":"New Wide-Net-Casting Jailbreak Attacks Risk Large Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19792","citing_title":"Mechanisms of Object Localization in Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2512.17321","citing_title":"Neuro-Symbolic Control with Large Language Models for Language-Guided Spatial Tasks","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23024","citing_title":"InCoM: Intent-Driven Perception and Structured Coordination for Mobile Manipulation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27437","citing_title":"SpatialStack: Layered Geometry-Language Fusion for 3D VLM Spatial Reasoning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03231","citing_title":"CoME-VL: Scaling Complementary Multi-Encoder Vision-Language Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10977","citing_title":"PASA: A Principled Embedding-Space Watermarking Approach for LLM-Generated Text under Semantic-Invariant Attacks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22557","citing_title":"Are Natural-Domain Foundation Models Effective for Accelerated Cardiac MRI Reconstruction?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01687","citing_title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07812","citing_title":"HAWK: Head Importance-Aware Visual Token Pruning in Multimodal Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14710","citing_title":"G-MIXER: Geodesic Mixup-based Implicit Semantic Expansion and Explicit Semantic Re-ranking for Zero-Shot Composed Image Retrieval","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17174","citing_title":"Modeling Multi-Dimensional Cognitive States in Large Language Models under Cognitive Crowding","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG","json":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG.json","graph_json":"https://pith.science/api/pith-number/QZC3UH5JBJ3XKV2VBIHEMH7BXG/graph.json","events_json":"https://pith.science/api/pith-number/QZC3UH5JBJ3XKV2VBIHEMH7BXG/events.json","paper":"https://pith.science/paper/QZC3UH5J"},"agent_actions":{"view_html":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG","download_json":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG.json","view_paper":"https://pith.science/paper/QZC3UH5J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.08825&json=true","fetch_graph":"https://pith.science/api/pith-number/QZC3UH5JBJ3XKV2VBIHEMH7BXG/graph.json","fetch_events":"https://pith.science/api/pith-number/QZC3UH5JBJ3XKV2VBIHEMH7BXG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG/action/storage_attestation","attest_author":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG/action/author_attestation","sign_citation":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG/action/citation_signature","submit_replication":"https://pith.science/pith/QZC3UH5JBJ3XKV2VBIHEMH7BXG/action/replication_record"}},"created_at":"2026-07-05T07:53:31.519200+00:00","updated_at":"2026-07-05T07:53:31.519200+00:00"}