{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JOTBVBRLQDT25ULSG5VO5W4BNF","short_pith_number":"pith:JOTBVBRL","schema_version":"1.0","canonical_sha256":"4ba61a862b80e7aed172376aeedb81694878cab1e110cdf49a1fc4eab1be549c","source":{"kind":"arxiv","id":"2410.15732","version":2},"attestation_state":"computed","paper":{"title":"ViMoE: An Empirical Study of Designing Vision Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenhui Qiang, Longhui Wei, Qi Tian, Xin He, Xumeng Han, Yingfei Sun, Zhenjun Han, Zhiyang Dou, Zipeng Wang","submitted_at":"2024-10-21T07:51:17Z","abstract_excerpt":"Mixture-of-Experts (MoE) models embody the divide-and-conquer concept and are a promising approach for increasing model capacity, demonstrating excellent scalability across multiple domains. In this paper, we integrate the MoE structure into the classic Vision Transformer (ViT), naming it ViMoE, and explore the potential of applying MoE to vision through a comprehensive study on image classification and semantic segmentation. However, we observe that the performance is sensitive to the configuration of MoE layers, making it challenging to obtain optimal results without careful design. The unde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15732","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-21T07:51:17Z","cross_cats_sorted":[],"title_canon_sha256":"dbb7457499eef48c4f45f617a1d7a6c6608f0c08d818704d7c274ce568f47760","abstract_canon_sha256":"adf91808b9c55b6423e71d99da260b18eb1d8c32d8c543ecfaa23b83eeaf87a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:40:20.279533Z","signature_b64":"BnpfnZnZdN+6jnlrOLckefcq6aYYDvny10D7t149vgP8nJzd65JMePr5C85eVvPJPkAOBWa7w7utreq2ocuLDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ba61a862b80e7aed172376aeedb81694878cab1e110cdf49a1fc4eab1be549c","last_reissued_at":"2026-07-05T09:40:20.279094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:40:20.279094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViMoE: An Empirical Study of Designing Vision Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenhui Qiang, Longhui Wei, Qi Tian, Xin He, Xumeng Han, Yingfei Sun, Zhenjun Han, Zhiyang Dou, Zipeng Wang","submitted_at":"2024-10-21T07:51:17Z","abstract_excerpt":"Mixture-of-Experts (MoE) models embody the divide-and-conquer concept and are a promising approach for increasing model capacity, demonstrating excellent scalability across multiple domains. In this paper, we integrate the MoE structure into the classic Vision Transformer (ViT), naming it ViMoE, and explore the potential of applying MoE to vision through a comprehensive study on image classification and semantic segmentation. However, we observe that the performance is sensitive to the configuration of MoE layers, making it challenging to obtain optimal results without careful design. The unde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15732","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15732/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15732","created_at":"2026-07-05T09:40:20.279150+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15732v2","created_at":"2026-07-05T09:40:20.279150+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15732","created_at":"2026-07-05T09:40:20.279150+00:00"},{"alias_kind":"pith_short_12","alias_value":"JOTBVBRLQDT2","created_at":"2026-07-05T09:40:20.279150+00:00"},{"alias_kind":"pith_short_16","alias_value":"JOTBVBRLQDT25ULS","created_at":"2026-07-05T09:40:20.279150+00:00"},{"alias_kind":"pith_short_8","alias_value":"JOTBVBRL","created_at":"2026-07-05T09:40:20.279150+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.32040","citing_title":"FaceMoE: Mixture of Experts for Low-Resolution Face Recognition","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31201","citing_title":"ExPLoRe: Expert Patch-Level Loss Routing for Multi-Objective Masked Image Modeling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20610","citing_title":"Beyond Routing: Characterising Expert Tuning and Representation in Vision Mixture-of-Experts","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15484","citing_title":"When Does Sparse MoE Help in Vision? The Role of Backbone Compute Leverage in Sparse Routing","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11232","citing_title":"DoReMi: Bridging 3D Domains via Topology-Aware Domain-Representation Mixture of Experts","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13761","citing_title":"Design and Behavior of Sparse Mixture-of-Experts Layers in CNN-based Semantic Segmentation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02641","citing_title":"Mamoda2.5: Enhancing Unified Multimodal Model with DiT-MoE","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF","json":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF.json","graph_json":"https://pith.science/api/pith-number/JOTBVBRLQDT25ULSG5VO5W4BNF/graph.json","events_json":"https://pith.science/api/pith-number/JOTBVBRLQDT25ULSG5VO5W4BNF/events.json","paper":"https://pith.science/paper/JOTBVBRL"},"agent_actions":{"view_html":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF","download_json":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF.json","view_paper":"https://pith.science/paper/JOTBVBRL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15732&json=true","fetch_graph":"https://pith.science/api/pith-number/JOTBVBRLQDT25ULSG5VO5W4BNF/graph.json","fetch_events":"https://pith.science/api/pith-number/JOTBVBRLQDT25ULSG5VO5W4BNF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF/action/storage_attestation","attest_author":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF/action/author_attestation","sign_citation":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF/action/citation_signature","submit_replication":"https://pith.science/pith/JOTBVBRLQDT25ULSG5VO5W4BNF/action/replication_record"}},"created_at":"2026-07-05T09:40:20.279150+00:00","updated_at":"2026-07-05T09:40:20.279150+00:00"}