{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QM2AKLQ4PLSRAORLD2HFHDR5OR","short_pith_number":"pith:QM2AKLQ4","schema_version":"1.0","canonical_sha256":"8334052e1c7ae5103a2b1e8e538e3d74581b6908af22c909aff7d70b2e7b0efb","source":{"kind":"arxiv","id":"2406.11832","version":2},"attestation_state":"computed","paper":{"title":"Unveiling Encoder-Free Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Haiwen Diao, Huchuan Lu, Xiaotong Li, Xinlong Wang, Yueze Wang, Yufeng Cui","submitted_at":"2024-06-17T17:59:44Z","abstract_excerpt":"Existing vision-language models (VLMs) mostly rely on vision encoders to extract visual features followed by large language models (LLMs) for visual-language tasks. However, the vision encoders set a strong inductive bias in abstracting visual representation, e.g., resolution, aspect ratio, and semantic priors, which could impede the flexibility and efficiency of the VLMs. Training pure VLMs that accept the seamless vision and language inputs, i.e., without vision encoders, remains challenging and rarely explored. Empirical observations reveal that direct training without encoders results in s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11832","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-17T17:59:44Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"87208fc42ac897e55a14962f6a17a16e809161f805fd06e0747df1f3f890e3c4","abstract_canon_sha256":"aad977884969bb81986a90ca5a7ecf4f25e1a81eb3887119dfad0dec3378cf46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:23.171535Z","signature_b64":"3rHCr4JxcaJMKJPXIIqo+tqduMMrPFoAb3e1IiYgvHWmWomRoDqfEYrSl15+6qJlHoHmap91Y0YnVpFNRe+XBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8334052e1c7ae5103a2b1e8e538e3d74581b6908af22c909aff7d70b2e7b0efb","last_reissued_at":"2026-07-05T09:27:23.171061Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:23.171061Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unveiling Encoder-Free Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Haiwen Diao, Huchuan Lu, Xiaotong Li, Xinlong Wang, Yueze Wang, Yufeng Cui","submitted_at":"2024-06-17T17:59:44Z","abstract_excerpt":"Existing vision-language models (VLMs) mostly rely on vision encoders to extract visual features followed by large language models (LLMs) for visual-language tasks. However, the vision encoders set a strong inductive bias in abstracting visual representation, e.g., resolution, aspect ratio, and semantic priors, which could impede the flexibility and efficiency of the VLMs. Training pure VLMs that accept the seamless vision and language inputs, i.e., without vision encoders, remains challenging and rarely explored. Empirical observations reveal that direct training without encoders results in s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11832","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11832","created_at":"2026-07-05T09:27:23.171120+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11832v2","created_at":"2026-07-05T09:27:23.171120+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11832","created_at":"2026-07-05T09:27:23.171120+00:00"},{"alias_kind":"pith_short_12","alias_value":"QM2AKLQ4PLSR","created_at":"2026-07-05T09:27:23.171120+00:00"},{"alias_kind":"pith_short_16","alias_value":"QM2AKLQ4PLSRAORL","created_at":"2026-07-05T09:27:23.171120+00:00"},{"alias_kind":"pith_short_8","alias_value":"QM2AKLQ4","created_at":"2026-07-05T09:27:23.171120+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28820","citing_title":"From Pixels to Words -- Towards Native One-Vision Models at Scale","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14169","citing_title":"Autoregressive Video Generation without Vector Quantization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15564","citing_title":"Show-o2: Improved Native Unified Multimodal Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01844","citing_title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07726","citing_title":"PaliGemma: A versatile 3B VLM for transfer","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18869","citing_title":"Emu3: Next-Token Prediction is All You Need","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09088","citing_title":"Memory-Efficient Transfer Learning with Fading Side Networks via Masked Dual Path Distillation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04058","citing_title":"MP-ISMoE: Mixed-Precision Interactive Side Mixture-of-Experts for Efficient Transfer Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR","json":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR.json","graph_json":"https://pith.science/api/pith-number/QM2AKLQ4PLSRAORLD2HFHDR5OR/graph.json","events_json":"https://pith.science/api/pith-number/QM2AKLQ4PLSRAORLD2HFHDR5OR/events.json","paper":"https://pith.science/paper/QM2AKLQ4"},"agent_actions":{"view_html":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR","download_json":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR.json","view_paper":"https://pith.science/paper/QM2AKLQ4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11832&json=true","fetch_graph":"https://pith.science/api/pith-number/QM2AKLQ4PLSRAORLD2HFHDR5OR/graph.json","fetch_events":"https://pith.science/api/pith-number/QM2AKLQ4PLSRAORLD2HFHDR5OR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR/action/storage_attestation","attest_author":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR/action/author_attestation","sign_citation":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR/action/citation_signature","submit_replication":"https://pith.science/pith/QM2AKLQ4PLSRAORLD2HFHDR5OR/action/replication_record"}},"created_at":"2026-07-05T09:27:23.171120+00:00","updated_at":"2026-07-05T09:27:23.171120+00:00"}