{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:N7PGIIWXLROXDM6BJ67BCVIN7O","short_pith_number":"pith:N7PGIIWX","schema_version":"1.0","canonical_sha256":"6fde6422d75c5d71b3c14fbe11550dfb83f57fac249a0b5c47017e70700c543e","source":{"kind":"arxiv","id":"2306.08586","version":3},"attestation_state":"computed","paper":{"title":"Learning to Specialize: Joint Gating-Expert Training for Adaptive MoEs in Decentralized Settings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Ahmed Hassan Awadallah, Anastasios Kyrillidis, Chen Dun, Dimitrios Dimitriadis, Fangshuo Liao, Guoqing Zheng, Hamza ElMokhtar Shili, Mirian Hipolito Garcia, Robert Sim, Yehya Farhat","submitted_at":"2023-06-14T15:47:52Z","abstract_excerpt":"Mixture-of-Experts (MoEs) achieve scalability by dynamically activating subsets of their components. Yet, understanding how expertise emerges through joint training of gating mechanisms and experts remains incomplete, especially in scenarios without clear task partitions. Motivated by inference costs and data heterogeneity, we study how joint training of gating functions and experts can dynamically allocate domain-specific expertise across multiple underlying data distributions. As an outcome of our framework, we develop an instance tailored specifically to decentralized training scenarios, in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.08586","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-14T15:47:52Z","cross_cats_sorted":["cs.AI","math.OC"],"title_canon_sha256":"a431e35284b5ed52228e25153312618eabea55138482c980bc4722506194377c","abstract_canon_sha256":"0368e37b7e3583f1ae7311d3effb498c63d8f4c3c9d39f356992a047083df760"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:45.559117Z","signature_b64":"vZMUHvVfglvgmFCONT2i7pLalZvwXWMwwHtdrb1oqG3YAzPni/QyosBxcRHnwtkc0ZYhxggDH/0G2MHRjHlaAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6fde6422d75c5d71b3c14fbe11550dfb83f57fac249a0b5c47017e70700c543e","last_reissued_at":"2026-07-05T11:14:45.558661Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:45.558661Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning to Specialize: Joint Gating-Expert Training for Adaptive MoEs in Decentralized Settings","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC"],"primary_cat":"cs.LG","authors_text":"Ahmed Hassan Awadallah, Anastasios Kyrillidis, Chen Dun, Dimitrios Dimitriadis, Fangshuo Liao, Guoqing Zheng, Hamza ElMokhtar Shili, Mirian Hipolito Garcia, Robert Sim, Yehya Farhat","submitted_at":"2023-06-14T15:47:52Z","abstract_excerpt":"Mixture-of-Experts (MoEs) achieve scalability by dynamically activating subsets of their components. Yet, understanding how expertise emerges through joint training of gating mechanisms and experts remains incomplete, especially in scenarios without clear task partitions. Motivated by inference costs and data heterogeneity, we study how joint training of gating functions and experts can dynamically allocate domain-specific expertise across multiple underlying data distributions. As an outcome of our framework, we develop an instance tailored specifically to decentralized training scenarios, in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.08586","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.08586/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.08586","created_at":"2026-07-05T11:14:45.558715+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.08586v3","created_at":"2026-07-05T11:14:45.558715+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.08586","created_at":"2026-07-05T11:14:45.558715+00:00"},{"alias_kind":"pith_short_12","alias_value":"N7PGIIWXLROX","created_at":"2026-07-05T11:14:45.558715+00:00"},{"alias_kind":"pith_short_16","alias_value":"N7PGIIWXLROXDM6B","created_at":"2026-07-05T11:14:45.558715+00:00"},{"alias_kind":"pith_short_8","alias_value":"N7PGIIWX","created_at":"2026-07-05T11:14:45.558715+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24985","citing_title":"Retrieval-Augmented Personalization with Foundation Models for Wearable Stress Detection","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23070","citing_title":"FLEX-MoE: Federated Mixture-of-Experts with Load-balanced Expert Assignment for Edge Computing","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21264","citing_title":"FedCoE: Bridging Generalization and Personalization via Federated Coordinated Dual-level MoEs","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O","json":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O.json","graph_json":"https://pith.science/api/pith-number/N7PGIIWXLROXDM6BJ67BCVIN7O/graph.json","events_json":"https://pith.science/api/pith-number/N7PGIIWXLROXDM6BJ67BCVIN7O/events.json","paper":"https://pith.science/paper/N7PGIIWX"},"agent_actions":{"view_html":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O","download_json":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O.json","view_paper":"https://pith.science/paper/N7PGIIWX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.08586&json=true","fetch_graph":"https://pith.science/api/pith-number/N7PGIIWXLROXDM6BJ67BCVIN7O/graph.json","fetch_events":"https://pith.science/api/pith-number/N7PGIIWXLROXDM6BJ67BCVIN7O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O/action/storage_attestation","attest_author":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O/action/author_attestation","sign_citation":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O/action/citation_signature","submit_replication":"https://pith.science/pith/N7PGIIWXLROXDM6BJ67BCVIN7O/action/replication_record"}},"created_at":"2026-07-05T11:14:45.558715+00:00","updated_at":"2026-07-05T11:14:45.558715+00:00"}