{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LUU4NGDJYCFK4A4QTHVPEZKG3M","short_pith_number":"pith:LUU4NGDJ","schema_version":"1.0","canonical_sha256":"5d29c69869c08aae039099eaf26546db093bf6121b61e136c8f87da12500f2ed","source":{"kind":"arxiv","id":"2410.07524","version":2},"attestation_state":"computed","paper":{"title":"Upcycling Large Language Models into Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhinav Khattar, Ashwath Aithal, Bryan Catanzaro, Ethan He, Mohammad Shoeybi, Ryan Prenger, Shiqing Fan, Tong Liu, Vijay Korthikanti, Zijie Yan","submitted_at":"2024-10-10T01:36:03Z","abstract_excerpt":"Upcycling pre-trained dense language models into sparse mixture-of-experts (MoE) models is an efficient approach to increase the model capacity of already trained models. However, optimal techniques for upcycling at scale remain unclear. In this work, we conduct an extensive study of upcycling methods and hyperparameters for billion-parameter scale language models. We propose a novel \"virtual group\" initialization scheme and weight scaling approach to enable upcycling into fine-grained MoE architectures. Through ablations, we find that upcycling outperforms continued dense model training. In a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07524","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-10T01:36:03Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"71acd7b0c7e3c19ade3d4443e21543799523a60e8cbb7858ba625f601bb7d9bc","abstract_canon_sha256":"7e8c808f788e551fb5920ec128e1bc6a1baaa9e41548ebe07b3ba35a72ceb608"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:36.784668Z","signature_b64":"VNCY9X1nY0mA9QBTuN4599q+tkEH/fVCqzbS7OJ+WsGIhczqqnXeGQjcTTaukAMhsw/6XQkn/7jAI9HQmEc5BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d29c69869c08aae039099eaf26546db093bf6121b61e136c8f87da12500f2ed","last_reissued_at":"2026-07-05T11:21:36.784186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:36.784186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Upcycling Large Language Models into Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhinav Khattar, Ashwath Aithal, Bryan Catanzaro, Ethan He, Mohammad Shoeybi, Ryan Prenger, Shiqing Fan, Tong Liu, Vijay Korthikanti, Zijie Yan","submitted_at":"2024-10-10T01:36:03Z","abstract_excerpt":"Upcycling pre-trained dense language models into sparse mixture-of-experts (MoE) models is an efficient approach to increase the model capacity of already trained models. However, optimal techniques for upcycling at scale remain unclear. In this work, we conduct an extensive study of upcycling methods and hyperparameters for billion-parameter scale language models. We propose a novel \"virtual group\" initialization scheme and weight scaling approach to enable upcycling into fine-grained MoE architectures. Through ablations, we find that upcycling outperforms continued dense model training. In a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07524","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07524/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07524","created_at":"2026-07-05T11:21:36.784251+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07524v2","created_at":"2026-07-05T11:21:36.784251+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07524","created_at":"2026-07-05T11:21:36.784251+00:00"},{"alias_kind":"pith_short_12","alias_value":"LUU4NGDJYCFK","created_at":"2026-07-05T11:21:36.784251+00:00"},{"alias_kind":"pith_short_16","alias_value":"LUU4NGDJYCFK4A4Q","created_at":"2026-07-05T11:21:36.784251+00:00"},{"alias_kind":"pith_short_8","alias_value":"LUU4NGDJ","created_at":"2026-07-05T11:21:36.784251+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27431","citing_title":"Tackling Multimodal Learning Challenges with Mixture-of-Expert: A Survey","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12119","citing_title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08008","citing_title":"Beyond Sunk Costs: Boosting LLM Pre-training Efficiency via Orthogonal Growth of Mixture-of-Experts","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05276","citing_title":"SpikingBrain: Spiking Brain-inspired Large Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M","json":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M.json","graph_json":"https://pith.science/api/pith-number/LUU4NGDJYCFK4A4QTHVPEZKG3M/graph.json","events_json":"https://pith.science/api/pith-number/LUU4NGDJYCFK4A4QTHVPEZKG3M/events.json","paper":"https://pith.science/paper/LUU4NGDJ"},"agent_actions":{"view_html":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M","download_json":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M.json","view_paper":"https://pith.science/paper/LUU4NGDJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07524&json=true","fetch_graph":"https://pith.science/api/pith-number/LUU4NGDJYCFK4A4QTHVPEZKG3M/graph.json","fetch_events":"https://pith.science/api/pith-number/LUU4NGDJYCFK4A4QTHVPEZKG3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M/action/storage_attestation","attest_author":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M/action/author_attestation","sign_citation":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M/action/citation_signature","submit_replication":"https://pith.science/pith/LUU4NGDJYCFK4A4QTHVPEZKG3M/action/replication_record"}},"created_at":"2026-07-05T11:21:36.784251+00:00","updated_at":"2026-07-05T11:21:36.784251+00:00"}