{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KTJMRKWMJWZR4D7YVXK7QID5MW","short_pith_number":"pith:KTJMRKWM","schema_version":"1.0","canonical_sha256":"54d2c8aacc4db31e0ff8add5f8207d659aed9bf898dc100b7a9c5b586d153162","source":{"kind":"arxiv","id":"2402.12750","version":2},"attestation_state":"computed","paper":{"title":"Model Composition for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chi Chen, Fei Huang, Fuwen Luo, Ji Zhang, Maosong Sun, Ming Yan, Peng Li, Yang Liu, Yiyang Du, Zheng Fang, Ziyue Wang","submitted_at":"2024-02-20T06:38:10Z","abstract_excerpt":"Recent developments in Multimodal Large Language Models (MLLMs) have shown rapid progress, moving towards the goal of creating versatile MLLMs that understand inputs from various modalities. However, existing methods typically rely on joint training with paired multimodal instruction data, which is resource-intensive and challenging to extend to new modalities. In this paper, we propose a new paradigm through the model composition of existing MLLMs to create a new model that retains the modal understanding capabilities of each original model. Our basic implementation, NaiveMC, demonstrates the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12750","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-20T06:38:10Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"c175d150847bb7dcfebfb85701a0b25841cce145cbf3605e79b24d2164186085","abstract_canon_sha256":"6a1864aa6181afcafb4c9c7387062acaef274e4ddb200db23c6ca164f771ee04"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:41.507811Z","signature_b64":"A2c7sSya6ksBY4Xv8IVBNy7lzysXr0BYgDBmVExhtY0MspOTCSDW9m8rmroEKB/FNGeneWCH9d1dMt/sDyrrAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54d2c8aacc4db31e0ff8add5f8207d659aed9bf898dc100b7a9c5b586d153162","last_reissued_at":"2026-07-05T08:48:41.507389Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:41.507389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Composition for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chi Chen, Fei Huang, Fuwen Luo, Ji Zhang, Maosong Sun, Ming Yan, Peng Li, Yang Liu, Yiyang Du, Zheng Fang, Ziyue Wang","submitted_at":"2024-02-20T06:38:10Z","abstract_excerpt":"Recent developments in Multimodal Large Language Models (MLLMs) have shown rapid progress, moving towards the goal of creating versatile MLLMs that understand inputs from various modalities. However, existing methods typically rely on joint training with paired multimodal instruction data, which is resource-intensive and challenging to extend to new modalities. In this paper, we propose a new paradigm through the model composition of existing MLLMs to create a new model that retains the modal understanding capabilities of each original model. Our basic implementation, NaiveMC, demonstrates the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12750","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12750/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12750","created_at":"2026-07-05T08:48:41.507447+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12750v2","created_at":"2026-07-05T08:48:41.507447+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12750","created_at":"2026-07-05T08:48:41.507447+00:00"},{"alias_kind":"pith_short_12","alias_value":"KTJMRKWMJWZR","created_at":"2026-07-05T08:48:41.507447+00:00"},{"alias_kind":"pith_short_16","alias_value":"KTJMRKWMJWZR4D7Y","created_at":"2026-07-05T08:48:41.507447+00:00"},{"alias_kind":"pith_short_8","alias_value":"KTJMRKWM","created_at":"2026-07-05T08:48:41.507447+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22823","citing_title":"PivotMerge: Bridging Heterogeneous Multimodal Pre-training via Post-Alignment Model Merging","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW","json":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW.json","graph_json":"https://pith.science/api/pith-number/KTJMRKWMJWZR4D7YVXK7QID5MW/graph.json","events_json":"https://pith.science/api/pith-number/KTJMRKWMJWZR4D7YVXK7QID5MW/events.json","paper":"https://pith.science/paper/KTJMRKWM"},"agent_actions":{"view_html":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW","download_json":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW.json","view_paper":"https://pith.science/paper/KTJMRKWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12750&json=true","fetch_graph":"https://pith.science/api/pith-number/KTJMRKWMJWZR4D7YVXK7QID5MW/graph.json","fetch_events":"https://pith.science/api/pith-number/KTJMRKWMJWZR4D7YVXK7QID5MW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW/action/storage_attestation","attest_author":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW/action/author_attestation","sign_citation":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW/action/citation_signature","submit_replication":"https://pith.science/pith/KTJMRKWMJWZR4D7YVXK7QID5MW/action/replication_record"}},"created_at":"2026-07-05T08:48:41.507447+00:00","updated_at":"2026-07-05T08:48:41.507447+00:00"}