{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7QLHQ4WXEW3GBJV3ZYA6Z5XR3H","short_pith_number":"pith:7QLHQ4WX","schema_version":"1.0","canonical_sha256":"fc167872d725b660a6bbce01ecf6f1d9f2e24be26209d76bbd992b22a73f5960","source":{"kind":"arxiv","id":"2403.07816","version":1},"attestation_state":"computed","paper":{"title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baptiste Rozi\\`ere, Daniel Li, Hu Xu, Jacob Kahn, Jason Weston, Olga Golovneva, Sainbayar Sukhbaatar, Vasu Sharma, Wen-tau Yih, Xian Li, Xi Victoria Lin","submitted_at":"2024-03-12T16:54:58Z","abstract_excerpt":"We investigate efficient methods for training Large Language Models (LLMs) to possess capabilities in multiple specialized domains, such as coding, math reasoning and world knowledge. Our method, named Branch-Train-MiX (BTX), starts from a seed model, which is branched to train experts in embarrassingly parallel fashion with high throughput and reduced communication cost. After individual experts are asynchronously trained, BTX brings together their feedforward parameters as experts in Mixture-of-Expert (MoE) layers and averages the remaining parameters, followed by an MoE-finetuning stage to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.07816","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-12T16:54:58Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"33ccc20270f6c267683e11f26f635ffb2b5d6686e604de6951266405492a58cc","abstract_canon_sha256":"652511d4648e3519d85445484d3c8e629788cb53590303e7f48a7fd1866630d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:15.925242Z","signature_b64":"ZGCCqJ9lVQDpRLTby3VK/8g0Ac9bEwxaA0QNFkgKn6OSk6Iz6rP+qHGyMAlQfPb7kvZo16oCzreQCIJwfAtoDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc167872d725b660a6bbce01ecf6f1d9f2e24be26209d76bbd992b22a73f5960","last_reissued_at":"2026-07-05T07:55:15.924753Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:15.924753Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baptiste Rozi\\`ere, Daniel Li, Hu Xu, Jacob Kahn, Jason Weston, Olga Golovneva, Sainbayar Sukhbaatar, Vasu Sharma, Wen-tau Yih, Xian Li, Xi Victoria Lin","submitted_at":"2024-03-12T16:54:58Z","abstract_excerpt":"We investigate efficient methods for training Large Language Models (LLMs) to possess capabilities in multiple specialized domains, such as coding, math reasoning and world knowledge. Our method, named Branch-Train-MiX (BTX), starts from a seed model, which is branched to train experts in embarrassingly parallel fashion with high throughput and reduced communication cost. After individual experts are asynchronously trained, BTX brings together their feedforward parameters as experts in Mixture-of-Expert (MoE) layers and averages the remaining parameters, followed by an MoE-finetuning stage to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.07816","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.07816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.07816","created_at":"2026-07-05T07:55:15.924813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.07816v1","created_at":"2026-07-05T07:55:15.924813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.07816","created_at":"2026-07-05T07:55:15.924813+00:00"},{"alias_kind":"pith_short_12","alias_value":"7QLHQ4WXEW3G","created_at":"2026-07-05T07:55:15.924813+00:00"},{"alias_kind":"pith_short_16","alias_value":"7QLHQ4WXEW3GBJV3","created_at":"2026-07-05T07:55:15.924813+00:00"},{"alias_kind":"pith_short_8","alias_value":"7QLHQ4WX","created_at":"2026-07-05T07:55:15.924813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24722","citing_title":"Decentralised AI Training and Inference with BlockTrain","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06902","citing_title":"TALAN: Task-Aligned Latent Adaptation Networks for Targeted Post-Training of Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30518","citing_title":"Regime-Aware Peer Specialization for Robust RAG under Heterogeneous Knowledge Conflicts","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30486","citing_title":"Graph-Conditioned Mixture of Graph Neural Network Experts for Traffic Forecasting","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04996","citing_title":"Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07666","citing_title":"Model Merging in LLMs, MLLMs, and Beyond: Methods, Theories, Applications and Opportunities","ref_index":199,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13247","citing_title":"EMO: Frustratingly Easy Progressive Training of Extendable MoE","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13247","citing_title":"EMO: Frustratingly Easy Progressive Training of Extendable MoE","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13421","citing_title":"Combining pre-trained models via localized model averaging","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12476","citing_title":"Routers Learn the Geometry of Their Experts: Geometric Coupling in Sparse Mixture-of-Experts","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19016","citing_title":"AlignCultura: Towards Culturally Aligned Large Language Models?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18473","citing_title":"Train Separately, Merge Together: Modular Post-Training with Mixture-of-Experts","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22849","citing_title":"R$^3$AG: Retriever Routing for Retrieval-Augmented Generation","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H","json":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H.json","graph_json":"https://pith.science/api/pith-number/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/graph.json","events_json":"https://pith.science/api/pith-number/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/events.json","paper":"https://pith.science/paper/7QLHQ4WX"},"agent_actions":{"view_html":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H","download_json":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H.json","view_paper":"https://pith.science/paper/7QLHQ4WX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.07816&json=true","fetch_graph":"https://pith.science/api/pith-number/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/graph.json","fetch_events":"https://pith.science/api/pith-number/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/action/storage_attestation","attest_author":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/action/author_attestation","sign_citation":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/action/citation_signature","submit_replication":"https://pith.science/pith/7QLHQ4WXEW3GBJV3ZYA6Z5XR3H/action/replication_record"}},"created_at":"2026-07-05T07:55:15.924813+00:00","updated_at":"2026-07-05T07:55:15.924813+00:00"}