{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JQIYV4L2KXNGO23ZRARGWECZ2W","short_pith_number":"pith:JQIYV4L2","schema_version":"1.0","canonical_sha256":"4c118af17a55da676b7988226b1059d59521f5c97bba404881296b62e693b595","source":{"kind":"arxiv","id":"2408.05211","version":3},"attestation_state":"computed","paper":{"title":"VITA: Towards Open-Source Interactive Omni Multimodal LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Caifeng Shan, Chaoyou Fu, Di Yin, Haojia Lin, Haoyu Cao, Long Ma, Meng Zhao, Ran He, Rongrong Ji, Shaoqi Dong, Xiawu Zheng, Xing Sun, Xiong Wang, Yangze Li, Yi-Fan Zhang, Yuhang Dai, Yunhang Shen, Yunsheng Wu, Zuwei Long","submitted_at":"2024-08-09T17:59:49Z","abstract_excerpt":"The remarkable multimodal capabilities and interactive experience of GPT-4o underscore their necessity in practical applications, yet open-source models rarely excel in both areas. In this paper, we introduce VITA, the first-ever open-source Multimodal Large Language Model (MLLM) adept at simultaneous processing and analysis of Video, Image, Text, and Audio modalities, and meanwhile has an advanced multimodal interactive experience. Starting from Mixtral 8x7B as a language foundation, we expand its Chinese vocabulary followed by bilingual instruction tuning. We further endow the language model"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.05211","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-09T17:59:49Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"f7a795c5027e2dbf086e3746c0e60e1fc76f0a9aca306b60b869172b87e243a6","abstract_canon_sha256":"c22484306c0db67666708a9d20c75e816d5fd1d515ac590c9cf80e73e5e3bab7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:19.186251Z","signature_b64":"ax5lQ2nqqjeD3lKB19g9H90EuAQtB8WB6MV20WreZFWHawzuYx+CV72Z+MPk45c0P3l+y0rJe2xck+NhjP9fCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c118af17a55da676b7988226b1059d59521f5c97bba404881296b62e693b595","last_reissued_at":"2026-07-05T11:12:19.185753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:19.185753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VITA: Towards Open-Source Interactive Omni Multimodal LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Caifeng Shan, Chaoyou Fu, Di Yin, Haojia Lin, Haoyu Cao, Long Ma, Meng Zhao, Ran He, Rongrong Ji, Shaoqi Dong, Xiawu Zheng, Xing Sun, Xiong Wang, Yangze Li, Yi-Fan Zhang, Yuhang Dai, Yunhang Shen, Yunsheng Wu, Zuwei Long","submitted_at":"2024-08-09T17:59:49Z","abstract_excerpt":"The remarkable multimodal capabilities and interactive experience of GPT-4o underscore their necessity in practical applications, yet open-source models rarely excel in both areas. In this paper, we introduce VITA, the first-ever open-source Multimodal Large Language Model (MLLM) adept at simultaneous processing and analysis of Video, Image, Text, and Audio modalities, and meanwhile has an advanced multimodal interactive experience. Starting from Mixtral 8x7B as a language foundation, we expand its Chinese vocabulary followed by bilingual instruction tuning. We further endow the language model"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.05211","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.05211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.05211","created_at":"2026-07-05T11:12:19.185813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.05211v3","created_at":"2026-07-05T11:12:19.185813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.05211","created_at":"2026-07-05T11:12:19.185813+00:00"},{"alias_kind":"pith_short_12","alias_value":"JQIYV4L2KXNG","created_at":"2026-07-05T11:12:19.185813+00:00"},{"alias_kind":"pith_short_16","alias_value":"JQIYV4L2KXNGO23Z","created_at":"2026-07-05T11:12:19.185813+00:00"},{"alias_kind":"pith_short_8","alias_value":"JQIYV4L2","created_at":"2026-07-05T11:12:19.185813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06540","citing_title":"Hierarchical Acoustic-Semantic Modeling: Modality Separation and Semantic Coherence for Full-Duplex SLMs","ref_index":51,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01667","citing_title":"Temporal and Cross-Modal Alignment for Enhanced Audiovisual Video Captioning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00387","citing_title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26584","citing_title":"O-MARC: Omni Memory-Augmented Compression Distillation for Efficient Video Understanding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11167","citing_title":"Multi-Faceted Interactivity Alignment in Full-Duplex Speech Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10016","citing_title":"Training-Free Multimodal Large Language Model Orchestration","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18758","citing_title":"OmniGUI: Benchmarking GUI Agents in Omni-Modal Smartphone Environments","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21008","citing_title":"A Survey of Audio Reasoning in Multimodal Foundation Models","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17360","citing_title":"Omni-DuplexEval: Evaluating Real-time Duplex Omni-modal Interaction","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10016","citing_title":"Training-Free Multimodal Large Language Model Orchestration","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2501.01957","citing_title":"VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14582","citing_title":"OmniZip: Audio-Guided Dynamic Token Compression for Fast Omnimodal Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02231","citing_title":"See, Hear, and Understand: Benchmarking Audiovisual Human Speech Understanding in Multimodal Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17196","citing_title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05271","citing_title":"DeepEyesV2: Toward Agentic Multimodal Model","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2409.17146","citing_title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12056","citing_title":"OmniRefine: Alignment-Aware Cooperative Compression for Efficient Omnimodal Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10199","citing_title":"How Should LLMs Listen While Speaking? A Study of User-Stream Routing in Full-Duplex Spoken Dialogue","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03937","citing_title":"MiniMind-O Technical Report: An Open Small-Scale Speech-Native Omni Model","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01024","citing_title":"EmoMM: Benchmarking and Steering MLLM for Multimodal Emotion Recognition under Conflict and Missingness","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19221","citing_title":"UAF: A Unified Audio Front-end LLM for Full-Duplex Speech Interaction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07593","citing_title":"TraceAV-Bench: Benchmarking Multi-Hop Trajectory Reasoning over Long Audio-Visual Videos","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14951","citing_title":"RaTA-Tool: Retrieval-based Tool Selection with Multimodal Large Language Models","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W","json":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W.json","graph_json":"https://pith.science/api/pith-number/JQIYV4L2KXNGO23ZRARGWECZ2W/graph.json","events_json":"https://pith.science/api/pith-number/JQIYV4L2KXNGO23ZRARGWECZ2W/events.json","paper":"https://pith.science/paper/JQIYV4L2"},"agent_actions":{"view_html":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W","download_json":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W.json","view_paper":"https://pith.science/paper/JQIYV4L2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.05211&json=true","fetch_graph":"https://pith.science/api/pith-number/JQIYV4L2KXNGO23ZRARGWECZ2W/graph.json","fetch_events":"https://pith.science/api/pith-number/JQIYV4L2KXNGO23ZRARGWECZ2W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W/action/storage_attestation","attest_author":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W/action/author_attestation","sign_citation":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W/action/citation_signature","submit_replication":"https://pith.science/pith/JQIYV4L2KXNGO23ZRARGWECZ2W/action/replication_record"}},"created_at":"2026-07-05T11:12:19.185813+00:00","updated_at":"2026-07-05T11:12:19.185813+00:00"}