{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HSEB325XZBDSL4J3GQXL3HML2O","short_pith_number":"pith:HSEB325X","schema_version":"1.0","canonical_sha256":"3c881debb7c84725f13b342ebd9d8bd3a4ceec96013bc18235739b01196c8cd1","source":{"kind":"arxiv","id":"2405.11273","version":1},"attestation_state":"computed","paper":{"title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.MM"],"primary_cat":"cs.AI","authors_text":"Baotian Hu, Lin Ma, Longyue Wang, Min Zhang, Shenyuan Jiang, Wanqi Zhong, Wenhan Luo, Yunxin Li","submitted_at":"2024-05-18T12:16:01Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) underscore the significance of scalable models and data to boost performance, yet this often incurs substantial computational costs. Although the Mixture of Experts (MoE) architecture has been employed to efficiently scale large language and image-text models, these efforts typically involve fewer experts and limited modalities. To address this, our work presents the pioneering attempt to develop a unified MLLM with the MoE architecture, named Uni-MoE that can handle a wide array of modalities. Specifically, it features modality-s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.11273","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-18T12:16:01Z","cross_cats_sorted":["cs.CL","cs.CV","cs.MM"],"title_canon_sha256":"65801dc3b4503c27f3bb6fdf64533588f75b11a531e3b30351bf34c5409691ff","abstract_canon_sha256":"b1f131311227ab15942bb0e437311c83e57750f77765359b7cb8a7ab7b8ae279"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:36.239119Z","signature_b64":"OXBo7mn4aFEpi5InjSUbKnMk39AAztdtiwhF+5VxiKkevtLPogEzSnS1zUHpq46nZxJzevBzGZJHNqKHfycqCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c881debb7c84725f13b342ebd9d8bd3a4ceec96013bc18235739b01196c8cd1","last_reissued_at":"2026-07-05T08:20:36.238695Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:36.238695Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.CV","cs.MM"],"primary_cat":"cs.AI","authors_text":"Baotian Hu, Lin Ma, Longyue Wang, Min Zhang, Shenyuan Jiang, Wanqi Zhong, Wenhan Luo, Yunxin Li","submitted_at":"2024-05-18T12:16:01Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) underscore the significance of scalable models and data to boost performance, yet this often incurs substantial computational costs. Although the Mixture of Experts (MoE) architecture has been employed to efficiently scale large language and image-text models, these efforts typically involve fewer experts and limited modalities. To address this, our work presents the pioneering attempt to develop a unified MLLM with the MoE architecture, named Uni-MoE that can handle a wide array of modalities. Specifically, it features modality-s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.11273","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.11273/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.11273","created_at":"2026-07-05T08:20:36.238750+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.11273v1","created_at":"2026-07-05T08:20:36.238750+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.11273","created_at":"2026-07-05T08:20:36.238750+00:00"},{"alias_kind":"pith_short_12","alias_value":"HSEB325XZBDS","created_at":"2026-07-05T08:20:36.238750+00:00"},{"alias_kind":"pith_short_16","alias_value":"HSEB325XZBDSL4J3","created_at":"2026-07-05T08:20:36.238750+00:00"},{"alias_kind":"pith_short_8","alias_value":"HSEB325X","created_at":"2026-07-05T08:20:36.238750+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29425","citing_title":"Mixture of Debaters: Learn to Debate at Architectural Level in Multi-Agent Reasoning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O","json":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O.json","graph_json":"https://pith.science/api/pith-number/HSEB325XZBDSL4J3GQXL3HML2O/graph.json","events_json":"https://pith.science/api/pith-number/HSEB325XZBDSL4J3GQXL3HML2O/events.json","paper":"https://pith.science/paper/HSEB325X"},"agent_actions":{"view_html":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O","download_json":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O.json","view_paper":"https://pith.science/paper/HSEB325X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.11273&json=true","fetch_graph":"https://pith.science/api/pith-number/HSEB325XZBDSL4J3GQXL3HML2O/graph.json","fetch_events":"https://pith.science/api/pith-number/HSEB325XZBDSL4J3GQXL3HML2O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O/action/storage_attestation","attest_author":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O/action/author_attestation","sign_citation":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O/action/citation_signature","submit_replication":"https://pith.science/pith/HSEB325XZBDSL4J3GQXL3HML2O/action/replication_record"}},"created_at":"2026-07-05T08:20:36.238750+00:00","updated_at":"2026-07-05T08:20:36.238750+00:00"}