{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:34EBOULNQXY6VJ344XYSTWZJMU","short_pith_number":"pith:34EBOULN","schema_version":"1.0","canonical_sha256":"df0817516d85f1eaa77ce5f129db29650d622883c87e3e85008709e968ff03ea","source":{"kind":"arxiv","id":"2204.07689","version":1},"attestation_state":"computed","paper":{"title":"Sparsely Activated Mixture-of-Experts are Robust Multi-Task Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ahmed H. Awadallah, Damien Jose, Eduardo Gonzalez, Jianfeng Gao, Krishan Subudhi, Shashank Gupta, Subhabrata Mukherjee","submitted_at":"2022-04-16T00:56:12Z","abstract_excerpt":"Traditional multi-task learning (MTL) methods use dense networks that use the same set of shared weights across several different tasks. This often creates interference where two or more tasks compete to pull model parameters in different directions. In this work, we study whether sparsely activated Mixture-of-Experts (MoE) improve multi-task learning by specializing some weights for learning shared representations and using the others for learning task-specific information. To this end, we devise task-aware gating functions to route examples from different tasks to specialized experts which s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.07689","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-16T00:56:12Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"68110828347eb5d76d9f502f832bd087f78e8da0ae98f574ba209d733b80b103","abstract_canon_sha256":"56b91be1d1dc4b8b89a45cb502497737c6c8f5445377486ffb430d6b26699258"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:15:17.483516Z","signature_b64":"B7Dab0K8ppN3cyaFLdTof1LyZkO+KYOVRjjnyKC/E559300/kNN2cCXtOVN2etX3ryTgi2untRNppLSFwX4fAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"df0817516d85f1eaa77ce5f129db29650d622883c87e3e85008709e968ff03ea","last_reissued_at":"2026-07-05T04:15:17.483088Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:15:17.483088Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparsely Activated Mixture-of-Experts are Robust Multi-Task Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ahmed H. Awadallah, Damien Jose, Eduardo Gonzalez, Jianfeng Gao, Krishan Subudhi, Shashank Gupta, Subhabrata Mukherjee","submitted_at":"2022-04-16T00:56:12Z","abstract_excerpt":"Traditional multi-task learning (MTL) methods use dense networks that use the same set of shared weights across several different tasks. This often creates interference where two or more tasks compete to pull model parameters in different directions. In this work, we study whether sparsely activated Mixture-of-Experts (MoE) improve multi-task learning by specializing some weights for learning shared representations and using the others for learning task-specific information. To this end, we devise task-aware gating functions to route examples from different tasks to specialized experts which s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.07689","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.07689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.07689","created_at":"2026-07-05T04:15:17.483147+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.07689v1","created_at":"2026-07-05T04:15:17.483147+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.07689","created_at":"2026-07-05T04:15:17.483147+00:00"},{"alias_kind":"pith_short_12","alias_value":"34EBOULNQXY6","created_at":"2026-07-05T04:15:17.483147+00:00"},{"alias_kind":"pith_short_16","alias_value":"34EBOULNQXY6VJ34","created_at":"2026-07-05T04:15:17.483147+00:00"},{"alias_kind":"pith_short_8","alias_value":"34EBOULN","created_at":"2026-07-05T04:15:17.483147+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01784","citing_title":"MoRE: A Mixture-of-Experts-Based Task-Adaptive End-to-End Network for Multimodal MRI Reconstruction","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":205,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08470","citing_title":"Joint Learning using Mixture-of-Expert-Based Representation for Speech Enhancement and Robust Emotion Recognition","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09355","citing_title":"FLAME: Adaptive Mixture-of-Experts for Continual Multimodal Multi-Task Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18866","citing_title":"HMR-Net: Hierarchical Modular Routing for Cross-Domain Object Detection in Aerial Images","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU","json":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU.json","graph_json":"https://pith.science/api/pith-number/34EBOULNQXY6VJ344XYSTWZJMU/graph.json","events_json":"https://pith.science/api/pith-number/34EBOULNQXY6VJ344XYSTWZJMU/events.json","paper":"https://pith.science/paper/34EBOULN"},"agent_actions":{"view_html":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU","download_json":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU.json","view_paper":"https://pith.science/paper/34EBOULN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.07689&json=true","fetch_graph":"https://pith.science/api/pith-number/34EBOULNQXY6VJ344XYSTWZJMU/graph.json","fetch_events":"https://pith.science/api/pith-number/34EBOULNQXY6VJ344XYSTWZJMU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU/action/storage_attestation","attest_author":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU/action/author_attestation","sign_citation":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU/action/citation_signature","submit_replication":"https://pith.science/pith/34EBOULNQXY6VJ344XYSTWZJMU/action/replication_record"}},"created_at":"2026-07-05T04:15:17.483147+00:00","updated_at":"2026-07-05T04:15:17.483147+00:00"}