{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MFG5AZXPX7WUFMSBIY6XKXSFFQ","short_pith_number":"pith:MFG5AZXP","schema_version":"1.0","canonical_sha256":"614dd066efbfed42b241463d755e452c2c7d06b032b14a5152f50c3e7fe16220","source":{"kind":"arxiv","id":"2403.14520","version":4},"attestation_state":"computed","paper":{"title":"Cobra: Extending Mamba to Multi-Modal Large Language Model for Efficient Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Donglin Wang, Han Zhao, Min Zhang, Pengxiang Ding, Siteng Huang, Wei Zhao","submitted_at":"2024-03-21T16:17:57Z","abstract_excerpt":"In recent years, the application of multimodal large language models (MLLM) in various fields has achieved remarkable success. However, as the foundation model for many downstream tasks, current MLLMs are composed of the well-known Transformer network, which has a less efficient quadratic computation complexity. To improve the efficiency of such basic models, we propose Cobra, a linear computational complexity MLLM. Specifically, Cobra integrates the efficient Mamba language model into the visual modality. Moreover, we explore and study various modal fusion schemes to create an effective multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14520","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-21T16:17:57Z","cross_cats_sorted":[],"title_canon_sha256":"1b4c7cd93d0278edeb9593fcca7a0584968c71a6978d97d97ce4d32c347e0ab2","abstract_canon_sha256":"d563e2c7b6b665f28f8dd889935ba8a684a3d27413586d34054f2cea085c02f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:19.347021Z","signature_b64":"bHjXfE7ehZO+T6XDg/GWdTRKrGCAdzwl2CzAvoVfp57tz77K0Nc6JVhLyI4aAzo7hLGYLD9nKLokr992VWkiDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"614dd066efbfed42b241463d755e452c2c7d06b032b14a5152f50c3e7fe16220","last_reissued_at":"2026-07-05T09:58:19.346558Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:19.346558Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cobra: Extending Mamba to Multi-Modal Large Language Model for Efficient Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Donglin Wang, Han Zhao, Min Zhang, Pengxiang Ding, Siteng Huang, Wei Zhao","submitted_at":"2024-03-21T16:17:57Z","abstract_excerpt":"In recent years, the application of multimodal large language models (MLLM) in various fields has achieved remarkable success. However, as the foundation model for many downstream tasks, current MLLMs are composed of the well-known Transformer network, which has a less efficient quadratic computation complexity. To improve the efficiency of such basic models, we propose Cobra, a linear computational complexity MLLM. Specifically, Cobra integrates the efficient Mamba language model into the visual modality. Moreover, we explore and study various modal fusion schemes to create an effective multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14520","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14520/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14520","created_at":"2026-07-05T09:58:19.346615+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14520v4","created_at":"2026-07-05T09:58:19.346615+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14520","created_at":"2026-07-05T09:58:19.346615+00:00"},{"alias_kind":"pith_short_12","alias_value":"MFG5AZXPX7WU","created_at":"2026-07-05T09:58:19.346615+00:00"},{"alias_kind":"pith_short_16","alias_value":"MFG5AZXPX7WUFMSB","created_at":"2026-07-05T09:58:19.346615+00:00"},{"alias_kind":"pith_short_8","alias_value":"MFG5AZXP","created_at":"2026-07-05T09:58:19.346615+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23126","citing_title":"MambaADv2: Evolving Duality-enhanced State Space Model for Unsupervised Anomaly Detection","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00390","citing_title":"Zamba2-VL Technical Report","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01129","citing_title":"A Survey of Mamba","ref_index":238,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ","json":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ.json","graph_json":"https://pith.science/api/pith-number/MFG5AZXPX7WUFMSBIY6XKXSFFQ/graph.json","events_json":"https://pith.science/api/pith-number/MFG5AZXPX7WUFMSBIY6XKXSFFQ/events.json","paper":"https://pith.science/paper/MFG5AZXP"},"agent_actions":{"view_html":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ","download_json":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ.json","view_paper":"https://pith.science/paper/MFG5AZXP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14520&json=true","fetch_graph":"https://pith.science/api/pith-number/MFG5AZXPX7WUFMSBIY6XKXSFFQ/graph.json","fetch_events":"https://pith.science/api/pith-number/MFG5AZXPX7WUFMSBIY6XKXSFFQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ/action/storage_attestation","attest_author":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ/action/author_attestation","sign_citation":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ/action/citation_signature","submit_replication":"https://pith.science/pith/MFG5AZXPX7WUFMSBIY6XKXSFFQ/action/replication_record"}},"created_at":"2026-07-05T09:58:19.346615+00:00","updated_at":"2026-07-05T09:58:19.346615+00:00"}