{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:M56GCB7RUNEBY4ZMGRMU2LUGXX","short_pith_number":"pith:M56GCB7R","schema_version":"1.0","canonical_sha256":"677c6107f1a3481c732c34594d2e86bde77ede78e6d0c89aa2362122472f7aad","source":{"kind":"arxiv","id":"2305.14705","version":2},"attestation_state":"computed","paper":{"title":"Mixture-of-Experts Meets Instruction Tuning:A Winning Combination for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Albert Webson, Barret Zoph, Denny Zhou, Hongkun Yu, Hyung Won Chung, Jason Wei, Kurt Keutzer, Le Hou, Nan Du, Shayne Longpre, Sheng Shen, Trevor Darrell, Tu Vu, Vincent Zhao, William Fedus, Wuyang Chen, Xinyun Chen, Yanqi Zhou, Yuexin Wu, Yunxuan Li","submitted_at":"2023-05-24T04:22:26Z","abstract_excerpt":"Sparse Mixture-of-Experts (MoE) is a neural architecture design that can be utilized to add learnable parameters to Large Language Models (LLMs) without increasing inference cost. Instruction tuning is a technique for training LLMs to follow instructions. We advocate combining these two approaches, as we find that MoE models benefit more from instruction tuning than dense models. In particular, we conduct empirical studies across three experimental setups: (i) Direct finetuning on individual downstream tasks devoid of instruction tuning; (ii) Instructiontuning followed by in-context few-shot o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14705","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T04:22:26Z","cross_cats_sorted":[],"title_canon_sha256":"5a866fc4ac06467798a7088bc64c608c72575a9ff8bb4ab04651d61a5eb0a4bb","abstract_canon_sha256":"ac4ea789b090f06fa071ad448e193fef3bcbd03822cf649d2edfd41a43db4ffa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:27:49.811223Z","signature_b64":"WNFxgyhzGTpW2xSdbCjmEWUgMn+o8nShpzDlRAGlcY0e705u4FOMmxUz249gjUF1byUseooRhhuZywLKWK7oCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"677c6107f1a3481c732c34594d2e86bde77ede78e6d0c89aa2362122472f7aad","last_reissued_at":"2026-07-05T06:27:49.810681Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:27:49.810681Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mixture-of-Experts Meets Instruction Tuning:A Winning Combination for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Albert Webson, Barret Zoph, Denny Zhou, Hongkun Yu, Hyung Won Chung, Jason Wei, Kurt Keutzer, Le Hou, Nan Du, Shayne Longpre, Sheng Shen, Trevor Darrell, Tu Vu, Vincent Zhao, William Fedus, Wuyang Chen, Xinyun Chen, Yanqi Zhou, Yuexin Wu, Yunxuan Li","submitted_at":"2023-05-24T04:22:26Z","abstract_excerpt":"Sparse Mixture-of-Experts (MoE) is a neural architecture design that can be utilized to add learnable parameters to Large Language Models (LLMs) without increasing inference cost. Instruction tuning is a technique for training LLMs to follow instructions. We advocate combining these two approaches, as we find that MoE models benefit more from instruction tuning than dense models. In particular, we conduct empirical studies across three experimental setups: (i) Direct finetuning on individual downstream tasks devoid of instruction tuning; (ii) Instructiontuning followed by in-context few-shot o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14705","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14705/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14705","created_at":"2026-07-05T06:27:49.810754+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14705v2","created_at":"2026-07-05T06:27:49.810754+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14705","created_at":"2026-07-05T06:27:49.810754+00:00"},{"alias_kind":"pith_short_12","alias_value":"M56GCB7RUNEB","created_at":"2026-07-05T06:27:49.810754+00:00"},{"alias_kind":"pith_short_16","alias_value":"M56GCB7RUNEBY4ZM","created_at":"2026-07-05T06:27:49.810754+00:00"},{"alias_kind":"pith_short_8","alias_value":"M56GCB7R","created_at":"2026-07-05T06:27:49.810754+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26287","citing_title":"GeMoE: Gating Entropy is All You Need for Uncertainty-aware Adaptive Routing in MoE-based Large Vision-Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2401.06066","citing_title":"DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19016","citing_title":"AlignCultura: Towards Culturally Aligned Large Language Models?","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07173","citing_title":"InfiniLoRA: Disaggregated Multi-LoRA Serving for Large Language Models","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX","json":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX.json","graph_json":"https://pith.science/api/pith-number/M56GCB7RUNEBY4ZMGRMU2LUGXX/graph.json","events_json":"https://pith.science/api/pith-number/M56GCB7RUNEBY4ZMGRMU2LUGXX/events.json","paper":"https://pith.science/paper/M56GCB7R"},"agent_actions":{"view_html":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX","download_json":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX.json","view_paper":"https://pith.science/paper/M56GCB7R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14705&json=true","fetch_graph":"https://pith.science/api/pith-number/M56GCB7RUNEBY4ZMGRMU2LUGXX/graph.json","fetch_events":"https://pith.science/api/pith-number/M56GCB7RUNEBY4ZMGRMU2LUGXX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX/action/storage_attestation","attest_author":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX/action/author_attestation","sign_citation":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX/action/citation_signature","submit_replication":"https://pith.science/pith/M56GCB7RUNEBY4ZMGRMU2LUGXX/action/replication_record"}},"created_at":"2026-07-05T06:27:49.810754+00:00","updated_at":"2026-07-05T06:27:49.810754+00:00"}