{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IDQGI542M2MEVHRHOSQ573RM6D","short_pith_number":"pith:IDQGI542","schema_version":"1.0","canonical_sha256":"40e064779a66984a9e2774a1dfee2cf0d4b82dd543af09594b27587fde74016e","source":{"kind":"arxiv","id":"2405.05949","version":1},"attestation_state":"computed","paper":{"title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chia-Wen Kuo, Fan Chen, Humphrey Shi, Jiachen Li, Jitesh Jain, Longyin Wen, Lu Xu, Sijie Zhu, Xinyao Wang","submitted_at":"2024-05-09T17:37:20Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (LLMs) have focused primarily on scaling by increasing text-image pair data and enhancing LLMs to improve performance on multimodal tasks. However, these scaling approaches are computationally expensive and overlook the significance of improving model capabilities from the vision side. Inspired by the successful applications of Mixture-of-Experts (MoE) in LLMs, which improves model scalability during training while keeping inference costs similar to those of smaller models, we propose CuMo. CuMo incorporates Co-upcycled Top-K sparsely-gat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.05949","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-09T17:37:20Z","cross_cats_sorted":[],"title_canon_sha256":"0b5614689ca599a5f87cd94369bf76d9e8babbcde9f999c0f6e76f6268541cc5","abstract_canon_sha256":"dabe60d5fd6741db929852a0af39704bcaedc706762a44b2b528954c3450b636"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:17:25.200115Z","signature_b64":"FBJ7L/d2HN9hnnfmwL7/qrkn1PS7XsJE2scL9jYzviWIDcr3LHEgcOou6QTcx/+AIN5T4ynHY9/hpsNkwHHODg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40e064779a66984a9e2774a1dfee2cf0d4b82dd543af09594b27587fde74016e","last_reissued_at":"2026-07-05T08:17:25.199629Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:17:25.199629Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chia-Wen Kuo, Fan Chen, Humphrey Shi, Jiachen Li, Jitesh Jain, Longyin Wen, Lu Xu, Sijie Zhu, Xinyao Wang","submitted_at":"2024-05-09T17:37:20Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (LLMs) have focused primarily on scaling by increasing text-image pair data and enhancing LLMs to improve performance on multimodal tasks. However, these scaling approaches are computationally expensive and overlook the significance of improving model capabilities from the vision side. Inspired by the successful applications of Mixture-of-Experts (MoE) in LLMs, which improves model scalability during training while keeping inference costs similar to those of smaller models, we propose CuMo. CuMo incorporates Co-upcycled Top-K sparsely-gat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.05949","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.05949/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.05949","created_at":"2026-07-05T08:17:25.199688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.05949v1","created_at":"2026-07-05T08:17:25.199688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.05949","created_at":"2026-07-05T08:17:25.199688+00:00"},{"alias_kind":"pith_short_12","alias_value":"IDQGI542M2ME","created_at":"2026-07-05T08:17:25.199688+00:00"},{"alias_kind":"pith_short_16","alias_value":"IDQGI542M2MEVHRH","created_at":"2026-07-05T08:17:25.199688+00:00"},{"alias_kind":"pith_short_8","alias_value":"IDQGI542","created_at":"2026-07-05T08:17:25.199688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00275","citing_title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D","json":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D.json","graph_json":"https://pith.science/api/pith-number/IDQGI542M2MEVHRHOSQ573RM6D/graph.json","events_json":"https://pith.science/api/pith-number/IDQGI542M2MEVHRHOSQ573RM6D/events.json","paper":"https://pith.science/paper/IDQGI542"},"agent_actions":{"view_html":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D","download_json":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D.json","view_paper":"https://pith.science/paper/IDQGI542","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.05949&json=true","fetch_graph":"https://pith.science/api/pith-number/IDQGI542M2MEVHRHOSQ573RM6D/graph.json","fetch_events":"https://pith.science/api/pith-number/IDQGI542M2MEVHRHOSQ573RM6D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D/action/storage_attestation","attest_author":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D/action/author_attestation","sign_citation":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D/action/citation_signature","submit_replication":"https://pith.science/pith/IDQGI542M2MEVHRHOSQ573RM6D/action/replication_record"}},"created_at":"2026-07-05T08:17:25.199688+00:00","updated_at":"2026-07-05T08:17:25.199688+00:00"}