{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BROPQPHU4HH4LULWXVRK7BBQHR","short_pith_number":"pith:BROPQPHU","schema_version":"1.0","canonical_sha256":"0c5cf83cf4e1cfc5d176bd62af84303c4f1404573bec7bb0d8ba2a6a3618a026","source":{"kind":"arxiv","id":"2501.04322","version":2},"attestation_state":"computed","paper":{"title":"Eve: Efficient Multimodal Vision Language Models with Elastic Visual Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuanjian Liu, Kai Han, Miao Rang, Yehui Tang, Yunhe Wang, Zhenni Bi","submitted_at":"2025-01-08T07:42:54Z","abstract_excerpt":"Multimodal vision language models (VLMs) have made significant progress with the support of continuously increasing model sizes and data volumes. Running VLMs on edge devices has become a challenge for their widespread application. There are several efficient VLM efforts, but they often sacrifice linguistic capabilities to enhance multimodal abilities, or require extensive training. To address this quandary,we introduce the innovative framework of Efficient Vision Language Models with Elastic Visual Experts (Eve). By strategically incorporating adaptable visual expertise at multiple stages of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.04322","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-08T07:42:54Z","cross_cats_sorted":[],"title_canon_sha256":"56ac385a537a92e34962194f8ac91a2bd25e517c47c9517f67c91dd0ffc99931","abstract_canon_sha256":"2d1a4c1ae07372caee6cc8e9c75a0ffc3c004eb0da9284545f0e78e0b5e5af6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:04:21.687114Z","signature_b64":"IQWNXv2SkmolvdIcepnYHyjSZaeSwnw8UMF2ZDPyu1AZMEP1+e7TgIFSdS2syPhBZ2nP5BsghQP7jSMt+GLJBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c5cf83cf4e1cfc5d176bd62af84303c4f1404573bec7bb0d8ba2a6a3618a026","last_reissued_at":"2026-07-05T10:04:21.686649Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:04:21.686649Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Eve: Efficient Multimodal Vision Language Models with Elastic Visual Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuanjian Liu, Kai Han, Miao Rang, Yehui Tang, Yunhe Wang, Zhenni Bi","submitted_at":"2025-01-08T07:42:54Z","abstract_excerpt":"Multimodal vision language models (VLMs) have made significant progress with the support of continuously increasing model sizes and data volumes. Running VLMs on edge devices has become a challenge for their widespread application. There are several efficient VLM efforts, but they often sacrifice linguistic capabilities to enhance multimodal abilities, or require extensive training. To address this quandary,we introduce the innovative framework of Efficient Vision Language Models with Elastic Visual Experts (Eve). By strategically incorporating adaptable visual expertise at multiple stages of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.04322","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.04322/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.04322","created_at":"2026-07-05T10:04:21.686707+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.04322v2","created_at":"2026-07-05T10:04:21.686707+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.04322","created_at":"2026-07-05T10:04:21.686707+00:00"},{"alias_kind":"pith_short_12","alias_value":"BROPQPHU4HH4","created_at":"2026-07-05T10:04:21.686707+00:00"},{"alias_kind":"pith_short_16","alias_value":"BROPQPHU4HH4LULW","created_at":"2026-07-05T10:04:21.686707+00:00"},{"alias_kind":"pith_short_8","alias_value":"BROPQPHU","created_at":"2026-07-05T10:04:21.686707+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10651","citing_title":"Kwai Keye-VL-2.0 Technical Report","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16416","citing_title":"Circle-RoPE: Cone-like Decoupled Rotary Positional Embedding for Large Vision-Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2501.04001","citing_title":"Sa2VA: Marrying SAM2 with LLaVA for Dense Grounded Understanding of Images and Videos","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR","json":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR.json","graph_json":"https://pith.science/api/pith-number/BROPQPHU4HH4LULWXVRK7BBQHR/graph.json","events_json":"https://pith.science/api/pith-number/BROPQPHU4HH4LULWXVRK7BBQHR/events.json","paper":"https://pith.science/paper/BROPQPHU"},"agent_actions":{"view_html":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR","download_json":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR.json","view_paper":"https://pith.science/paper/BROPQPHU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.04322&json=true","fetch_graph":"https://pith.science/api/pith-number/BROPQPHU4HH4LULWXVRK7BBQHR/graph.json","fetch_events":"https://pith.science/api/pith-number/BROPQPHU4HH4LULWXVRK7BBQHR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR/action/storage_attestation","attest_author":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR/action/author_attestation","sign_citation":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR/action/citation_signature","submit_replication":"https://pith.science/pith/BROPQPHU4HH4LULWXVRK7BBQHR/action/replication_record"}},"created_at":"2026-07-05T10:04:21.686707+00:00","updated_at":"2026-07-05T10:04:21.686707+00:00"}