{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZOTKHO63SMQMD4OZOPDRKECO53","short_pith_number":"pith:ZOTKHO63","schema_version":"1.0","canonical_sha256":"cba6a3bbdb9320c1f1d973c715104eeed36074c35745a47da36aebecde53bfe0","source":{"kind":"arxiv","id":"2402.02526","version":1},"attestation_state":"computed","paper":{"title":"CompeteSMoE -- Effective Training of Sparse Mixture of Experts via Competition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Binh T. Nguyen, Chenghao Liu, Giang Do, Huy Nguyen, Mina Sartipi, Nhat Ho, Quang Pham, Savitha Ramasamy, Steven Hoi, TrungTin Nguyen, Xiaoli Li","submitted_at":"2024-02-04T15:17:09Z","abstract_excerpt":"Sparse mixture of experts (SMoE) offers an appealing solution to scale up the model complexity beyond the mean of increasing the network's depth or width. However, effective training of SMoE has proven to be challenging due to the representation collapse issue, which causes parameter redundancy and limited representation potentials. In this work, we propose a competition mechanism to address this fundamental challenge of representation collapse. By routing inputs only to experts with the highest neural response, we show that, under mild assumptions, competition enjoys the same convergence rate"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.02526","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-04T15:17:09Z","cross_cats_sorted":[],"title_canon_sha256":"6c19ae49fa750cce7a080a2c2c78d57d6308afcf5f6942311cc833bb58610afd","abstract_canon_sha256":"d51ec733d0f931c96732d5b076aa2d17dbc36c2822e1b5a44239287ddff890b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:29.766103Z","signature_b64":"C8lHmCH3IK1e44w9z0LJATHRl/zfOEWqrpNlHM+3edqts6lwHIfAQ4IJ0RjfsqIO+cQ147BmSfOQhz2tshIjBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cba6a3bbdb9320c1f1d973c715104eeed36074c35745a47da36aebecde53bfe0","last_reissued_at":"2026-07-05T07:41:29.765564Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:29.765564Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CompeteSMoE -- Effective Training of Sparse Mixture of Experts via Competition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Binh T. Nguyen, Chenghao Liu, Giang Do, Huy Nguyen, Mina Sartipi, Nhat Ho, Quang Pham, Savitha Ramasamy, Steven Hoi, TrungTin Nguyen, Xiaoli Li","submitted_at":"2024-02-04T15:17:09Z","abstract_excerpt":"Sparse mixture of experts (SMoE) offers an appealing solution to scale up the model complexity beyond the mean of increasing the network's depth or width. However, effective training of SMoE has proven to be challenging due to the representation collapse issue, which causes parameter redundancy and limited representation potentials. In this work, we propose a competition mechanism to address this fundamental challenge of representation collapse. By routing inputs only to experts with the highest neural response, we show that, under mild assumptions, competition enjoys the same convergence rate"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.02526","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.02526/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.02526","created_at":"2026-07-05T07:41:29.765623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.02526v1","created_at":"2026-07-05T07:41:29.765623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.02526","created_at":"2026-07-05T07:41:29.765623+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZOTKHO63SMQM","created_at":"2026-07-05T07:41:29.765623+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZOTKHO63SMQMD4OZ","created_at":"2026-07-05T07:41:29.765623+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZOTKHO63","created_at":"2026-07-05T07:41:29.765623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31397","citing_title":"Mixture-of-Control: State-Aware Fine-Tuning for Transformer-based Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2502.15315","citing_title":"Tight Clusters Make Specialized Experts","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06845","citing_title":"Convergence Rates for Latent Mixing Measures in Infinite Homoscedastic Location-Scale Mixture Models","ref_index":95,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53","json":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53.json","graph_json":"https://pith.science/api/pith-number/ZOTKHO63SMQMD4OZOPDRKECO53/graph.json","events_json":"https://pith.science/api/pith-number/ZOTKHO63SMQMD4OZOPDRKECO53/events.json","paper":"https://pith.science/paper/ZOTKHO63"},"agent_actions":{"view_html":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53","download_json":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53.json","view_paper":"https://pith.science/paper/ZOTKHO63","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.02526&json=true","fetch_graph":"https://pith.science/api/pith-number/ZOTKHO63SMQMD4OZOPDRKECO53/graph.json","fetch_events":"https://pith.science/api/pith-number/ZOTKHO63SMQMD4OZOPDRKECO53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53/action/storage_attestation","attest_author":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53/action/author_attestation","sign_citation":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53/action/citation_signature","submit_replication":"https://pith.science/pith/ZOTKHO63SMQMD4OZOPDRKECO53/action/replication_record"}},"created_at":"2026-07-05T07:41:29.765623+00:00","updated_at":"2026-07-05T07:41:29.765623+00:00"}