{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RRYTNBHYYRNXNFOCU6EJSEJBWM","short_pith_number":"pith:RRYTNBHY","schema_version":"1.0","canonical_sha256":"8c713684f8c45b7695c2a788991121b30ed9d2009cefb6dc2fabeb2c8ee93d5b","source":{"kind":"arxiv","id":"2308.00951","version":2},"attestation_state":"computed","paper":{"title":"From Sparse to Soft Mixtures of Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Basil Mustafa, Carlos Riquelme, Joan Puigcerver, Neil Houlsby","submitted_at":"2023-08-02T05:20:55Z","abstract_excerpt":"Sparse mixture of expert architectures (MoEs) scale model capacity without significant increases in training or inference costs. Despite their success, MoEs suffer from a number of issues: training instability, token dropping, inability to scale the number of experts, or ineffective finetuning. In this work, we propose Soft MoE, a fully-differentiable sparse Transformer that addresses these challenges, while maintaining the benefits of MoEs. Soft MoE performs an implicit soft assignment by passing different weighted combinations of all input tokens to each expert. As in other MoEs, experts in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.00951","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-08-02T05:20:55Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"131502f3ef7beb573fa4ddf6c7a6512232a2a41c73f52cc5a476ebe92faf17b7","abstract_canon_sha256":"8c303c92a4d4841fb8b591d424b2dd77557a74b20f3494a31dcef568f86b5dfd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:11.053824Z","signature_b64":"2OjbroJqB/Zyqzo2vr0hsCivzTEiJm2QohtIxh8xKs/OaPaKOihoDVyd2tueYGBrL0DrbcG77wD9u/I2UfaMDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c713684f8c45b7695c2a788991121b30ed9d2009cefb6dc2fabeb2c8ee93d5b","last_reissued_at":"2026-07-05T08:23:11.053304Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:11.053304Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Sparse to Soft Mixtures of Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.LG","authors_text":"Basil Mustafa, Carlos Riquelme, Joan Puigcerver, Neil Houlsby","submitted_at":"2023-08-02T05:20:55Z","abstract_excerpt":"Sparse mixture of expert architectures (MoEs) scale model capacity without significant increases in training or inference costs. Despite their success, MoEs suffer from a number of issues: training instability, token dropping, inability to scale the number of experts, or ineffective finetuning. In this work, we propose Soft MoE, a fully-differentiable sparse Transformer that addresses these challenges, while maintaining the benefits of MoEs. Soft MoE performs an implicit soft assignment by passing different weighted combinations of all input tokens to each expert. As in other MoEs, experts in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.00951","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.00951/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.00951","created_at":"2026-07-05T08:23:11.053382+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.00951v2","created_at":"2026-07-05T08:23:11.053382+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.00951","created_at":"2026-07-05T08:23:11.053382+00:00"},{"alias_kind":"pith_short_12","alias_value":"RRYTNBHYYRNX","created_at":"2026-07-05T08:23:11.053382+00:00"},{"alias_kind":"pith_short_16","alias_value":"RRYTNBHYYRNXNFOC","created_at":"2026-07-05T08:23:11.053382+00:00"},{"alias_kind":"pith_short_8","alias_value":"RRYTNBHY","created_at":"2026-07-05T08:23:11.053382+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26287","citing_title":"GeMoE: Gating Entropy is All You Need for Uncertainty-aware Adaptive Routing in MoE-based Large Vision-Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32040","citing_title":"FaceMoE: Mixture of Experts for Low-Resolution Face Recognition","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27431","citing_title":"Tackling Multimodal Learning Challenges with Mixture-of-Expert: A Survey","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2502.15315","citing_title":"Tight Clusters Make Specialized Experts","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07287","citing_title":"SplatWeaver: Learning to Allocate Gaussian Primitives for Generalizable Novel View Synthesis","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18297","citing_title":"Path-Constrained Mixture-of-Experts","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.24245","citing_title":"B-MoE: A Body-Part-Aware Mixture-of-Experts \"All Parts Matter\" Approach to Micro-Action Recognition","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00539","citing_title":"AGoQ: Activation and Gradient Quantization for Memory-Efficient Distributed Training of LLMs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24661","citing_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00539","citing_title":"AGoQ: Activation and Gradient Quantization for Memory-Efficient Distributed Training of LLMs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04058","citing_title":"MP-ISMoE: Mixed-Precision Interactive Side Mixture-of-Experts for Efficient Transfer Learning","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07287","citing_title":"SplatWeaver: Learning to Allocate Gaussian Primitives for Generalizable Novel View Synthesis","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07260","citing_title":"When Are Experts Misrouted? Counterfactual Routing Analysis in Mixture-of-Experts Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24661","citing_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13761","citing_title":"Design and Behavior of Sparse Mixture-of-Experts Layers in CNN-based Semantic Segmentation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM","json":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM.json","graph_json":"https://pith.science/api/pith-number/RRYTNBHYYRNXNFOCU6EJSEJBWM/graph.json","events_json":"https://pith.science/api/pith-number/RRYTNBHYYRNXNFOCU6EJSEJBWM/events.json","paper":"https://pith.science/paper/RRYTNBHY"},"agent_actions":{"view_html":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM","download_json":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM.json","view_paper":"https://pith.science/paper/RRYTNBHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.00951&json=true","fetch_graph":"https://pith.science/api/pith-number/RRYTNBHYYRNXNFOCU6EJSEJBWM/graph.json","fetch_events":"https://pith.science/api/pith-number/RRYTNBHYYRNXNFOCU6EJSEJBWM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM/action/storage_attestation","attest_author":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM/action/author_attestation","sign_citation":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM/action/citation_signature","submit_replication":"https://pith.science/pith/RRYTNBHYYRNXNFOCU6EJSEJBWM/action/replication_record"}},"created_at":"2026-07-05T08:23:11.053382+00:00","updated_at":"2026-07-05T08:23:11.053382+00:00"}