{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:OSL5OBPUDZEXQKDT3VOJ4TBZAT","short_pith_number":"pith:OSL5OBPU","schema_version":"1.0","canonical_sha256":"7497d705f41e49782873dd5c9e4c3904ea7f7e9e88b5b26de9e2110791bc957a","source":{"kind":"arxiv","id":"2212.05055","version":2},"attestation_state":"computed","paper":{"title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Aran Komatsuzaki, Basil Mustafa, Carlos Riquelme Ruiz, James Lee-Thorp, Joan Puigcerver, Joshua Ainslie, Mostafa Dehghani, Neil Houlsby, Yi Tay","submitted_at":"2022-12-09T18:57:37Z","abstract_excerpt":"Training large, deep neural networks to convergence can be prohibitively expensive. As a result, often only a small selection of popular, dense models are reused across different contexts and tasks. Increasingly, sparsely activated models, which seek to decouple model size from computation costs, are becoming an attractive alternative to dense models. Although more efficient in terms of quality and computation cost, sparse models remain data-hungry and costly to train from scratch in the large scale regime. In this work, we propose sparse upcycling -- a simple way to reuse sunk training costs "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.05055","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-09T18:57:37Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"5c3f8e69669d60e011faf0a150ff78fca097efca7476bd997c0f57499acd6199","abstract_canon_sha256":"20b569b204e58e28c9ee2fc97213c31ddeaf4faf4917e670b8206b0920a766b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:43:03.849467Z","signature_b64":"gskquY+hxvdnjiF6YmqMQtBsSGtOsOGa/Xlw4w75Sdm3UFYmdjy/VKiK1aow4orF6cp8Zu82PEwx2zy20ovGBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7497d705f41e49782873dd5c9e4c3904ea7f7e9e88b5b26de9e2110791bc957a","last_reissued_at":"2026-07-05T05:43:03.849055Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:43:03.849055Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Aran Komatsuzaki, Basil Mustafa, Carlos Riquelme Ruiz, James Lee-Thorp, Joan Puigcerver, Joshua Ainslie, Mostafa Dehghani, Neil Houlsby, Yi Tay","submitted_at":"2022-12-09T18:57:37Z","abstract_excerpt":"Training large, deep neural networks to convergence can be prohibitively expensive. As a result, often only a small selection of popular, dense models are reused across different contexts and tasks. Increasingly, sparsely activated models, which seek to decouple model size from computation costs, are becoming an attractive alternative to dense models. Although more efficient in terms of quality and computation cost, sparse models remain data-hungry and costly to train from scratch in the large scale regime. In this work, we propose sparse upcycling -- a simple way to reuse sunk training costs "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.05055","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.05055/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.05055","created_at":"2026-07-05T05:43:03.849117+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.05055v2","created_at":"2026-07-05T05:43:03.849117+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.05055","created_at":"2026-07-05T05:43:03.849117+00:00"},{"alias_kind":"pith_short_12","alias_value":"OSL5OBPUDZEX","created_at":"2026-07-05T05:43:03.849117+00:00"},{"alias_kind":"pith_short_16","alias_value":"OSL5OBPUDZEXQKDT","created_at":"2026-07-05T05:43:03.849117+00:00"},{"alias_kind":"pith_short_8","alias_value":"OSL5OBPU","created_at":"2026-07-05T05:43:03.849117+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26287","citing_title":"GeMoE: Gating Entropy is All You Need for Uncertainty-aware Adaptive Routing in MoE-based Large Vision-Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21645","citing_title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24901","citing_title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10369","citing_title":"PADD: Path-Aligned Decompression Distillation for Non-Router Teacher to Guide MoE Student Learning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09038","citing_title":"Personalization Meets Safety:Mechanisms,Risks,and Mitigations in Personalized LLMs","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07404","citing_title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00293","citing_title":"Rosetta: Composable Native Multimodal Pretraining","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00275","citing_title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26496","citing_title":"Dense2MoE: Pushing the Pareto Frontier of On-Device LLMs via Unified Pruning and Upcycling","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04416","citing_title":"Analytical FFN-to-MoE Restructuring via Activation Pattern Analysis","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12119","citing_title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05276","citing_title":"SpikingBrain: Spiking Brain-inspired Large Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15947","citing_title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13997","citing_title":"HodgeCover: Higher-Order Topological Coverage Drives Compression of Sparse Mixture-of-Experts","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2404.06395","citing_title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2305.13245","citing_title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT","json":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT.json","graph_json":"https://pith.science/api/pith-number/OSL5OBPUDZEXQKDT3VOJ4TBZAT/graph.json","events_json":"https://pith.science/api/pith-number/OSL5OBPUDZEXQKDT3VOJ4TBZAT/events.json","paper":"https://pith.science/paper/OSL5OBPU"},"agent_actions":{"view_html":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT","download_json":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT.json","view_paper":"https://pith.science/paper/OSL5OBPU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.05055&json=true","fetch_graph":"https://pith.science/api/pith-number/OSL5OBPUDZEXQKDT3VOJ4TBZAT/graph.json","fetch_events":"https://pith.science/api/pith-number/OSL5OBPUDZEXQKDT3VOJ4TBZAT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT/action/storage_attestation","attest_author":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT/action/author_attestation","sign_citation":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT/action/citation_signature","submit_replication":"https://pith.science/pith/OSL5OBPUDZEXQKDT3VOJ4TBZAT/action/replication_record"}},"created_at":"2026-07-05T05:43:03.849117+00:00","updated_at":"2026-07-05T05:43:03.849117+00:00"}