{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:X3QCVLBJNITDVXVSCWWAKMDMPO","short_pith_number":"pith:X3QCVLBJ","schema_version":"1.0","canonical_sha256":"bee02aac296a263adeb215ac05306c7bb0a5dbda319f42f3cd373da2c8ecd74e","source":{"kind":"arxiv","id":"2110.04260","version":3},"attestation_state":"computed","paper":{"title":"Taming Sparsely Activated Transformer with Stochastic Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hany Hassan, Jianfeng Gao, Jian Jiao, Ruofei Zhang, Simiao Zuo, Tuo Zhao, XiaoDong Liu, Young Jin Kim","submitted_at":"2021-10-08T17:15:47Z","abstract_excerpt":"Sparsely activated models (SAMs), such as Mixture-of-Experts (MoE), can easily scale to have outrageously large amounts of parameters without significant increase in computational cost. However, SAMs are reported to be parameter inefficient such that larger models do not always lead to better performance. While most on-going research focuses on improving SAMs models by exploring methods of routing inputs to experts, our analysis reveals that such research might not lead to the solution we expect, i.e., the commonly-used routing methods based on gating mechanisms do not work better than randoml"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.04260","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-10-08T17:15:47Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"02fa24bf9d2842d3ce01f48b7dec51d1183cffd457388dc4b5e3bc1de22792da","abstract_canon_sha256":"8ed2553754247290e639eab48f5f801b0eb99fea5f949bc37d672b61683c1eeb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:54:03.007768Z","signature_b64":"GX86N8MgunzPnfbUsH/kdk6FqnrKzH2OvXzra/qy2W6gOxwD5qGyEmIZNIO4P4oy5wL1v9ap7MQlqQMaoyUCCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bee02aac296a263adeb215ac05306c7bb0a5dbda319f42f3cd373da2c8ecd74e","last_reissued_at":"2026-07-05T03:54:03.007280Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:54:03.007280Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taming Sparsely Activated Transformer with Stochastic Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hany Hassan, Jianfeng Gao, Jian Jiao, Ruofei Zhang, Simiao Zuo, Tuo Zhao, XiaoDong Liu, Young Jin Kim","submitted_at":"2021-10-08T17:15:47Z","abstract_excerpt":"Sparsely activated models (SAMs), such as Mixture-of-Experts (MoE), can easily scale to have outrageously large amounts of parameters without significant increase in computational cost. However, SAMs are reported to be parameter inefficient such that larger models do not always lead to better performance. While most on-going research focuses on improving SAMs models by exploring methods of routing inputs to experts, our analysis reveals that such research might not lead to the solution we expect, i.e., the commonly-used routing methods based on gating mechanisms do not work better than randoml"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.04260","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.04260/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.04260","created_at":"2026-07-05T03:54:03.007341+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.04260v3","created_at":"2026-07-05T03:54:03.007341+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.04260","created_at":"2026-07-05T03:54:03.007341+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3QCVLBJNITD","created_at":"2026-07-05T03:54:03.007341+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3QCVLBJNITDVXVS","created_at":"2026-07-05T03:54:03.007341+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3QCVLBJ","created_at":"2026-07-05T03:54:03.007341+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00371","citing_title":"MEPA: Multi-Scale Representation Alignment for Visual Autoregressive Modeling with Mixture of Experts","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2309.14509","citing_title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","ref_index":167,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO","json":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO.json","graph_json":"https://pith.science/api/pith-number/X3QCVLBJNITDVXVSCWWAKMDMPO/graph.json","events_json":"https://pith.science/api/pith-number/X3QCVLBJNITDVXVSCWWAKMDMPO/events.json","paper":"https://pith.science/paper/X3QCVLBJ"},"agent_actions":{"view_html":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO","download_json":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO.json","view_paper":"https://pith.science/paper/X3QCVLBJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.04260&json=true","fetch_graph":"https://pith.science/api/pith-number/X3QCVLBJNITDVXVSCWWAKMDMPO/graph.json","fetch_events":"https://pith.science/api/pith-number/X3QCVLBJNITDVXVSCWWAKMDMPO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO/action/storage_attestation","attest_author":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO/action/author_attestation","sign_citation":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO/action/citation_signature","submit_replication":"https://pith.science/pith/X3QCVLBJNITDVXVSCWWAKMDMPO/action/replication_record"}},"created_at":"2026-07-05T03:54:03.007341+00:00","updated_at":"2026-07-05T03:54:03.007341+00:00"}