{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:XI4MXGF3KDFIJIVZPRNVYBTZSG","short_pith_number":"pith:XI4MXGF3","schema_version":"1.0","canonical_sha256":"ba38cb98bb50ca84a2b97c5b5c067991b837667cde5d259bf50fa6dbb211ec63","source":{"kind":"arxiv","id":"2204.09636","version":3},"attestation_state":"computed","paper":{"title":"Residual Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongdong Chen, Lemeng Wu, Lu Yuan, Mengchen Liu, Xiyang Dai, Yinpeng Chen","submitted_at":"2022-04-20T17:29:48Z","abstract_excerpt":"Mixture of Experts (MoE) is able to scale up vision transformers effectively. However, it requires prohibiting computation resources to train a large MoE transformer. In this paper, we propose Residual Mixture of Experts (RMoE), an efficient training pipeline for MoE vision transformers on downstream tasks, such as segmentation and detection. RMoE achieves comparable results with the upper-bound MoE training, while only introducing minor additional training cost than the lower-bound non-MoE training pipelines. The efficiency is supported by our key observation: the weights of an MoE transforme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.09636","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-20T17:29:48Z","cross_cats_sorted":[],"title_canon_sha256":"276c12350b1ae846b13d26556b8b4e9aacd3ddb1aad9a1581cbb750b8cab5403","abstract_canon_sha256":"3da4641771918a035324238969a3b602e51fb8d2dc3dc1f9d2b0d5ee08cc79eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:03:00.253193Z","signature_b64":"ul6jdLiykrJRk8TtPvg87PuCHzB/EJl3qtW96MY0p+na9Nwb7sli4GCAamWFt8AAoVn5I3SmuwMNZhSnbB6sDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba38cb98bb50ca84a2b97c5b5c067991b837667cde5d259bf50fa6dbb211ec63","last_reissued_at":"2026-07-05T05:03:00.252746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:03:00.252746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Residual Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongdong Chen, Lemeng Wu, Lu Yuan, Mengchen Liu, Xiyang Dai, Yinpeng Chen","submitted_at":"2022-04-20T17:29:48Z","abstract_excerpt":"Mixture of Experts (MoE) is able to scale up vision transformers effectively. However, it requires prohibiting computation resources to train a large MoE transformer. In this paper, we propose Residual Mixture of Experts (RMoE), an efficient training pipeline for MoE vision transformers on downstream tasks, such as segmentation and detection. RMoE achieves comparable results with the upper-bound MoE training, while only introducing minor additional training cost than the lower-bound non-MoE training pipelines. The efficiency is supported by our key observation: the weights of an MoE transforme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.09636","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.09636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.09636","created_at":"2026-07-05T05:03:00.252805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.09636v3","created_at":"2026-07-05T05:03:00.252805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.09636","created_at":"2026-07-05T05:03:00.252805+00:00"},{"alias_kind":"pith_short_12","alias_value":"XI4MXGF3KDFI","created_at":"2026-07-05T05:03:00.252805+00:00"},{"alias_kind":"pith_short_16","alias_value":"XI4MXGF3KDFIJIVZ","created_at":"2026-07-05T05:03:00.252805+00:00"},{"alias_kind":"pith_short_8","alias_value":"XI4MXGF3","created_at":"2026-07-05T05:03:00.252805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25892","citing_title":"SP-MoMamba: Superpixel-driven Mixture of State Space Experts for Efficient Image Super-Resolution","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20610","citing_title":"Beyond Routing: Characterising Expert Tuning and Representation in Vision Mixture-of-Experts","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08209","citing_title":"Learngene Search Across Multiple Datasets for Building Variable-Sized Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG","json":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG.json","graph_json":"https://pith.science/api/pith-number/XI4MXGF3KDFIJIVZPRNVYBTZSG/graph.json","events_json":"https://pith.science/api/pith-number/XI4MXGF3KDFIJIVZPRNVYBTZSG/events.json","paper":"https://pith.science/paper/XI4MXGF3"},"agent_actions":{"view_html":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG","download_json":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG.json","view_paper":"https://pith.science/paper/XI4MXGF3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.09636&json=true","fetch_graph":"https://pith.science/api/pith-number/XI4MXGF3KDFIJIVZPRNVYBTZSG/graph.json","fetch_events":"https://pith.science/api/pith-number/XI4MXGF3KDFIJIVZPRNVYBTZSG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG/action/storage_attestation","attest_author":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG/action/author_attestation","sign_citation":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG/action/citation_signature","submit_replication":"https://pith.science/pith/XI4MXGF3KDFIJIVZPRNVYBTZSG/action/replication_record"}},"created_at":"2026-07-05T05:03:00.252805+00:00","updated_at":"2026-07-05T05:03:00.252805+00:00"}