{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RQO7OHEKS6UZIAMUD2ALGFZF3Z","short_pith_number":"pith:RQO7OHEK","schema_version":"1.0","canonical_sha256":"8c1df71c8a97a99401941e80b31725de74e74881e876002382f151d7b15629bb","source":{"kind":"arxiv","id":"2503.20680","version":1},"attestation_state":"computed","paper":{"title":"Vision as LoRA","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bingru Li, Can Huang, Han Wang, Jinghui Lu, Jingqun Tang, Yanjie Wang, YongJie Ye, Yuxiang Nie","submitted_at":"2025-03-26T16:15:42Z","abstract_excerpt":"We introduce Vision as LoRA (VoRA), a novel paradigm for transforming an LLM into an MLLM. Unlike prevalent MLLM architectures that rely on external vision modules for vision encoding, VoRA internalizes visual capabilities by integrating vision-specific LoRA layers directly into the LLM. This design allows the added parameters to be seamlessly merged into the LLM during inference, eliminating structural complexity and minimizing computational overhead. Moreover, inheriting the LLM's ability of handling flexible context, VoRA can process inputs at arbitrary resolutions.\n  To further strengthen "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20680","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-26T16:15:42Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"04a992b6ec48bfc67150be17ab6a90e1ee3a668281f3a4a3b058e23610715f14","abstract_canon_sha256":"63f352698756f624c00ade69da0f3dda20baafc5874c334fb69507cc7f90fb61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:42.364291Z","signature_b64":"80fekGos5mDWvXS4rfylFtGhIIL036DJYVL5X/BwkMA54+2k3LiSEjkQ7fDwR9v88XOkjcIGc6biJQadXs3FBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c1df71c8a97a99401941e80b31725de74e74881e876002382f151d7b15629bb","last_reissued_at":"2026-07-05T10:39:42.363713Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:42.363713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision as LoRA","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bingru Li, Can Huang, Han Wang, Jinghui Lu, Jingqun Tang, Yanjie Wang, YongJie Ye, Yuxiang Nie","submitted_at":"2025-03-26T16:15:42Z","abstract_excerpt":"We introduce Vision as LoRA (VoRA), a novel paradigm for transforming an LLM into an MLLM. Unlike prevalent MLLM architectures that rely on external vision modules for vision encoding, VoRA internalizes visual capabilities by integrating vision-specific LoRA layers directly into the LLM. This design allows the added parameters to be seamlessly merged into the LLM during inference, eliminating structural complexity and minimizing computational overhead. Moreover, inheriting the LLM's ability of handling flexible context, VoRA can process inputs at arbitrary resolutions.\n  To further strengthen "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20680","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20680/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20680","created_at":"2026-07-05T10:39:42.363782+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20680v1","created_at":"2026-07-05T10:39:42.363782+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20680","created_at":"2026-07-05T10:39:42.363782+00:00"},{"alias_kind":"pith_short_12","alias_value":"RQO7OHEKS6UZ","created_at":"2026-07-05T10:39:42.363782+00:00"},{"alias_kind":"pith_short_16","alias_value":"RQO7OHEKS6UZIAMU","created_at":"2026-07-05T10:39:42.363782+00:00"},{"alias_kind":"pith_short_8","alias_value":"RQO7OHEK","created_at":"2026-07-05T10:39:42.363782+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20077","citing_title":"The Hidden Evolution of Disguised Visual Context inside the VLM","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11033","citing_title":"AuRA: Internalizing Audio Understanding into LLMs as LoRA","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30189","citing_title":"DAIN: Dynamic Agent-Based Interaction Network for Efficient and Collaborative Multimodal Reasoning","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18173","citing_title":"Do You Need Text Rectification? Soft Attention Mask Embedding for Rectification-Free Scene Text Spotting","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19219","citing_title":"Selective LoRA for Visual Tokens and Attention Heads","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14548","citing_title":"Local Spatiotemporal Convolutional Network for Robust Gait Recognition","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03339","citing_title":"Hierarchical Awareness Adapters with Hybrid Pyramid Feature Fusion for Dense Depth Prediction","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12500","citing_title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18486","citing_title":"Xiaomi OneVL: One-Step Latent Reasoning and Planning with Vision-Language Explanation","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23750","citing_title":"The Override Gap: A Magnitude Account of Knowledge Conflict Failure in Hypernetwork-Based Instant LLM Adaptation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25188","citing_title":"Image Classification via Random Dilated Convolution with Multi-Branch Feature Extraction and Context Excitation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23750","citing_title":"The Override Gap: A Magnitude Account of Knowledge Conflict Failure in Hypernetwork-Based Instant LLM Adaptation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00885","citing_title":"Multi-Branch Non-Homogeneous Image Dehazing via Concentration Partitioning and Image Fusion","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18486","citing_title":"Xiaomi OneVL: One-Step Latent Reasoning and Planning with Vision-Language Explanation","ref_index":99,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z","json":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z.json","graph_json":"https://pith.science/api/pith-number/RQO7OHEKS6UZIAMUD2ALGFZF3Z/graph.json","events_json":"https://pith.science/api/pith-number/RQO7OHEKS6UZIAMUD2ALGFZF3Z/events.json","paper":"https://pith.science/paper/RQO7OHEK"},"agent_actions":{"view_html":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z","download_json":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z.json","view_paper":"https://pith.science/paper/RQO7OHEK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20680&json=true","fetch_graph":"https://pith.science/api/pith-number/RQO7OHEKS6UZIAMUD2ALGFZF3Z/graph.json","fetch_events":"https://pith.science/api/pith-number/RQO7OHEKS6UZIAMUD2ALGFZF3Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z/action/storage_attestation","attest_author":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z/action/author_attestation","sign_citation":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z/action/citation_signature","submit_replication":"https://pith.science/pith/RQO7OHEKS6UZIAMUD2ALGFZF3Z/action/replication_record"}},"created_at":"2026-07-05T10:39:42.363782+00:00","updated_at":"2026-07-05T10:39:42.363782+00:00"}