{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5D36O7SA7C4FQIVWWBVKV4PJBL","short_pith_number":"pith:5D36O7SA","schema_version":"1.0","canonical_sha256":"e8f7e77e40f8b85822b6b06aaaf1e90af722e3a83d88fe3de72ed712fa19fa3e","source":{"kind":"arxiv","id":"2204.08396","version":1},"attestation_state":"computed","paper":{"title":"StableMoE: Stable Routing Strategy for Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Baobao Chang, Bo Zheng, Damai Dai, Furu Wei, Li Dong, Shuming Ma, Zhifang Sui","submitted_at":"2022-04-18T16:48:19Z","abstract_excerpt":"The Mixture-of-Experts (MoE) technique can scale up the model size of Transformers with an affordable computational overhead. We point out that existing learning-to-route MoE methods suffer from the routing fluctuation issue, i.e., the target expert of the same input may change along with training, but only one expert will be activated for the input during inference. The routing fluctuation tends to harm sample efficiency because the same input updates different experts but only one is finally used. In this paper, we propose StableMoE with two training stages to address the routing fluctuation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.08396","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-04-18T16:48:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"eea6c0e6bcd55721b702653c69b7672c53e9255f6e9ff3e02652ef315f933d6b","abstract_canon_sha256":"b5669400ee069ae32402fd7fe58d3b921d5b43c61c1840d92dabebd2c7dbbc02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:15:25.446729Z","signature_b64":"rqGqrSJEZJCIZMscGwaSlEPMSQD5dqAFW+GQGQ0WE9I8HWyjM+Vzse7qe1vDFtv5jPaWZHQA42ECA1fmxTRNDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e8f7e77e40f8b85822b6b06aaaf1e90af722e3a83d88fe3de72ed712fa19fa3e","last_reissued_at":"2026-07-05T04:15:25.446197Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:15:25.446197Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StableMoE: Stable Routing Strategy for Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Baobao Chang, Bo Zheng, Damai Dai, Furu Wei, Li Dong, Shuming Ma, Zhifang Sui","submitted_at":"2022-04-18T16:48:19Z","abstract_excerpt":"The Mixture-of-Experts (MoE) technique can scale up the model size of Transformers with an affordable computational overhead. We point out that existing learning-to-route MoE methods suffer from the routing fluctuation issue, i.e., the target expert of the same input may change along with training, but only one expert will be activated for the input during inference. The routing fluctuation tends to harm sample efficiency because the same input updates different experts but only one is finally used. In this paper, we propose StableMoE with two training stages to address the routing fluctuation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.08396","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.08396/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.08396","created_at":"2026-07-05T04:15:25.446271+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.08396v1","created_at":"2026-07-05T04:15:25.446271+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.08396","created_at":"2026-07-05T04:15:25.446271+00:00"},{"alias_kind":"pith_short_12","alias_value":"5D36O7SA7C4F","created_at":"2026-07-05T04:15:25.446271+00:00"},{"alias_kind":"pith_short_16","alias_value":"5D36O7SA7C4FQIVW","created_at":"2026-07-05T04:15:25.446271+00:00"},{"alias_kind":"pith_short_8","alias_value":"5D36O7SA","created_at":"2026-07-05T04:15:25.446271+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17816","citing_title":"Conservation Laws for Modern Neural Architectures","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18297","citing_title":"Path-Constrained Mixture-of-Experts","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12476","citing_title":"Routers Learn the Geometry of Their Experts: Geometric Coupling in Sparse Mixture-of-Experts","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07260","citing_title":"When Are Experts Misrouted? Counterfactual Routing Analysis in Mixture-of-Experts Language Models","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL","json":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL.json","graph_json":"https://pith.science/api/pith-number/5D36O7SA7C4FQIVWWBVKV4PJBL/graph.json","events_json":"https://pith.science/api/pith-number/5D36O7SA7C4FQIVWWBVKV4PJBL/events.json","paper":"https://pith.science/paper/5D36O7SA"},"agent_actions":{"view_html":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL","download_json":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL.json","view_paper":"https://pith.science/paper/5D36O7SA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.08396&json=true","fetch_graph":"https://pith.science/api/pith-number/5D36O7SA7C4FQIVWWBVKV4PJBL/graph.json","fetch_events":"https://pith.science/api/pith-number/5D36O7SA7C4FQIVWWBVKV4PJBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL/action/storage_attestation","attest_author":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL/action/author_attestation","sign_citation":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL/action/citation_signature","submit_replication":"https://pith.science/pith/5D36O7SA7C4FQIVWWBVKV4PJBL/action/replication_record"}},"created_at":"2026-07-05T04:15:25.446271+00:00","updated_at":"2026-07-05T04:15:25.446271+00:00"}