{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X2XOGJTRYJDZXMPEAX76MLAC3H","short_pith_number":"pith:X2XOGJTR","schema_version":"1.0","canonical_sha256":"beaee32671c2479bb1e405ffe62c02d9d39cf8e61712e45c8aea63a09b1e6a26","source":{"kind":"arxiv","id":"2401.17221","version":1},"attestation_state":"computed","paper":{"title":"MouSi: Poly-Visual-Expert Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Boyang Hong, Caishuang Huang, Changhao Jiang, Guodong Zheng, Hang Yan, Junjie Ye, Junke Wang, Lu Chen, Ming Zhang, Qi Zhang, Rui Zheng, Senjie Jin, Shihan Dou, Shuo Li, Sirui Song, Tao Gui, Tao Ji, Xiaoran Fan, Xipeng Qiu, Xuanjing Huang, Yu-Gang Jiang, Yuhao Zhou, Zhiheng Xi, Zuxuan Wu","submitted_at":"2024-01-30T18:09:11Z","abstract_excerpt":"Current large vision-language models (VLMs) often encounter challenges such as insufficient capabilities of a single visual component and excessively long visual tokens. These issues can limit the model's effectiveness in accurately interpreting complex visual information and over-lengthy contextual information. Addressing these challenges is crucial for enhancing the performance and applicability of VLMs. This paper proposes the use of ensemble experts technique to synergizes the capabilities of individual visual encoders, including those skilled in image-text matching, OCR, image segmentatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.17221","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-30T18:09:11Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"b3500e363441936e72bd026cf8f9dd7a8884ba98bcf544c8294cb5c5da6fb495","abstract_canon_sha256":"e8ef4c4a69c4d4fc9f0d06530d336e728d9d7c9ad829eb7393c45c43e31e241c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:39:21.772178Z","signature_b64":"qXoRZnEQpLP8h/fR7vgVa6WrByx0vZsObIE4zX4W2LGkgIqD8OAnTvu1R0Mjb4cFuOMKhFOEQhwzAkzoioOECA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"beaee32671c2479bb1e405ffe62c02d9d39cf8e61712e45c8aea63a09b1e6a26","last_reissued_at":"2026-07-05T07:39:21.771678Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:39:21.771678Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MouSi: Poly-Visual-Expert Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Boyang Hong, Caishuang Huang, Changhao Jiang, Guodong Zheng, Hang Yan, Junjie Ye, Junke Wang, Lu Chen, Ming Zhang, Qi Zhang, Rui Zheng, Senjie Jin, Shihan Dou, Shuo Li, Sirui Song, Tao Gui, Tao Ji, Xiaoran Fan, Xipeng Qiu, Xuanjing Huang, Yu-Gang Jiang, Yuhao Zhou, Zhiheng Xi, Zuxuan Wu","submitted_at":"2024-01-30T18:09:11Z","abstract_excerpt":"Current large vision-language models (VLMs) often encounter challenges such as insufficient capabilities of a single visual component and excessively long visual tokens. These issues can limit the model's effectiveness in accurately interpreting complex visual information and over-lengthy contextual information. Addressing these challenges is crucial for enhancing the performance and applicability of VLMs. This paper proposes the use of ensemble experts technique to synergizes the capabilities of individual visual encoders, including those skilled in image-text matching, OCR, image segmentatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.17221","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.17221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.17221","created_at":"2026-07-05T07:39:21.771746+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.17221v1","created_at":"2026-07-05T07:39:21.771746+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.17221","created_at":"2026-07-05T07:39:21.771746+00:00"},{"alias_kind":"pith_short_12","alias_value":"X2XOGJTRYJDZ","created_at":"2026-07-05T07:39:21.771746+00:00"},{"alias_kind":"pith_short_16","alias_value":"X2XOGJTRYJDZXMPE","created_at":"2026-07-05T07:39:21.771746+00:00"},{"alias_kind":"pith_short_8","alias_value":"X2XOGJTR","created_at":"2026-07-05T07:39:21.771746+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00275","citing_title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14977","citing_title":"EchoVLM: Dynamic Mixture-of-Experts Vision-Language Model for Universal Ultrasound Intelligence","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H","json":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H.json","graph_json":"https://pith.science/api/pith-number/X2XOGJTRYJDZXMPEAX76MLAC3H/graph.json","events_json":"https://pith.science/api/pith-number/X2XOGJTRYJDZXMPEAX76MLAC3H/events.json","paper":"https://pith.science/paper/X2XOGJTR"},"agent_actions":{"view_html":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H","download_json":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H.json","view_paper":"https://pith.science/paper/X2XOGJTR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.17221&json=true","fetch_graph":"https://pith.science/api/pith-number/X2XOGJTRYJDZXMPEAX76MLAC3H/graph.json","fetch_events":"https://pith.science/api/pith-number/X2XOGJTRYJDZXMPEAX76MLAC3H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H/action/storage_attestation","attest_author":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H/action/author_attestation","sign_citation":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H/action/citation_signature","submit_replication":"https://pith.science/pith/X2XOGJTRYJDZXMPEAX76MLAC3H/action/replication_record"}},"created_at":"2026-07-05T07:39:21.771746+00:00","updated_at":"2026-07-05T07:39:21.771746+00:00"}