{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QFYC36FKDEG5A6X5QBFJ45PPXG","short_pith_number":"pith:QFYC36FK","schema_version":"1.0","canonical_sha256":"81702df8aa190dd07afd804a9e75efb9ac7ab4fc10705233349b3198471cee0b","source":{"kind":"arxiv","id":"2407.14177","version":1},"attestation_state":"computed","paper":{"title":"EVLM: An Efficient Vision-Language Model for Visual Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Wen, Changyi Liu, Dewen Fan, Di Xu, Di Zhang, Dong Shen, Fan Yang, Hanwen Zhong, Huasong Zhong, Huihui Xiao, Jiahong Wu, Kaibing Chen, Kui Xia, Size Li, Tianke Zhang, Wei Yuan, Yifei Hu","submitted_at":"2024-07-19T10:09:51Z","abstract_excerpt":"In the field of multi-modal language models, the majority of methods are built on an architecture similar to LLaVA. These models use a single-layer ViT feature as a visual prompt, directly feeding it into the language models alongside textual tokens. However, when dealing with long sequences of visual signals or inputs such as videos, the self-attention mechanism of language models can lead to significant computational overhead. Additionally, using single-layer ViT features makes it challenging for large language models to perceive visual signals fully. This paper proposes an efficient multi-m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.14177","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-19T10:09:51Z","cross_cats_sorted":[],"title_canon_sha256":"bb7b7c8e5fd607d822a46da25f001aaa15e9376c3310f9ab3b7e013e12ff516f","abstract_canon_sha256":"ce171d8d06fdead01ac004465b6152095905e38c7b1a24d26fadc80f8447685a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:59.546947Z","signature_b64":"JR/ifO+EY+N8cTogIgSn/53KejgyWTawcSoEFi/CqnwAVd6R4sJ/7WvOIwKRNUx1zrkr8R3YL8WfCfFUJTrTBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81702df8aa190dd07afd804a9e75efb9ac7ab4fc10705233349b3198471cee0b","last_reissued_at":"2026-07-05T08:45:59.546469Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:59.546469Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EVLM: An Efficient Vision-Language Model for Visual Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bin Wen, Changyi Liu, Dewen Fan, Di Xu, Di Zhang, Dong Shen, Fan Yang, Hanwen Zhong, Huasong Zhong, Huihui Xiao, Jiahong Wu, Kaibing Chen, Kui Xia, Size Li, Tianke Zhang, Wei Yuan, Yifei Hu","submitted_at":"2024-07-19T10:09:51Z","abstract_excerpt":"In the field of multi-modal language models, the majority of methods are built on an architecture similar to LLaVA. These models use a single-layer ViT feature as a visual prompt, directly feeding it into the language models alongside textual tokens. However, when dealing with long sequences of visual signals or inputs such as videos, the self-attention mechanism of language models can lead to significant computational overhead. Additionally, using single-layer ViT features makes it challenging for large language models to perceive visual signals fully. This paper proposes an efficient multi-m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.14177","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.14177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.14177","created_at":"2026-07-05T08:45:59.546530+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.14177v1","created_at":"2026-07-05T08:45:59.546530+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.14177","created_at":"2026-07-05T08:45:59.546530+00:00"},{"alias_kind":"pith_short_12","alias_value":"QFYC36FKDEG5","created_at":"2026-07-05T08:45:59.546530+00:00"},{"alias_kind":"pith_short_16","alias_value":"QFYC36FKDEG5A6X5","created_at":"2026-07-05T08:45:59.546530+00:00"},{"alias_kind":"pith_short_8","alias_value":"QFYC36FK","created_at":"2026-07-05T08:45:59.546530+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2408.04840","citing_title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","ref_index":202,"is_internal_anchor":false},{"citing_arxiv_id":"2409.17146","citing_title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG","json":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG.json","graph_json":"https://pith.science/api/pith-number/QFYC36FKDEG5A6X5QBFJ45PPXG/graph.json","events_json":"https://pith.science/api/pith-number/QFYC36FKDEG5A6X5QBFJ45PPXG/events.json","paper":"https://pith.science/paper/QFYC36FK"},"agent_actions":{"view_html":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG","download_json":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG.json","view_paper":"https://pith.science/paper/QFYC36FK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.14177&json=true","fetch_graph":"https://pith.science/api/pith-number/QFYC36FKDEG5A6X5QBFJ45PPXG/graph.json","fetch_events":"https://pith.science/api/pith-number/QFYC36FKDEG5A6X5QBFJ45PPXG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG/action/storage_attestation","attest_author":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG/action/author_attestation","sign_citation":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG/action/citation_signature","submit_replication":"https://pith.science/pith/QFYC36FKDEG5A6X5QBFJ45PPXG/action/replication_record"}},"created_at":"2026-07-05T08:45:59.546530+00:00","updated_at":"2026-07-05T08:45:59.546530+00:00"}