{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J6OGSFYRVUAD3VUI6ZYSBS66HC","short_pith_number":"pith:J6OGSFYR","schema_version":"1.0","canonical_sha256":"4f9c691711ad003dd688f67120cbde388f4c74a3e0d576df3b897a4834d88568","source":{"kind":"arxiv","id":"2506.18349","version":1},"attestation_state":"computed","paper":{"title":"SlimMoE: Structured Compression of Large MoE Models via Expert Slimming and Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chen Liang, Ilgee Hong, Tuo Zhao, Weizhu Chen, Young Jin Kim, Zichong Li, Zixuan Zhang","submitted_at":"2025-06-23T07:15:59Z","abstract_excerpt":"The Mixture of Experts (MoE) architecture has emerged as a powerful paradigm for scaling large language models (LLMs) while maintaining inference efficiency. However, their enormous memory requirements make them prohibitively expensive to fine-tune or deploy in resource-constrained environments. To address this challenge, we introduce SlimMoE, a multi-stage compression framework for transforming large MoE models into much smaller, efficient variants without incurring the prohibitive costs of training from scratch. Our method systematically reduces parameter counts by slimming experts and trans"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.18349","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-23T07:15:59Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8b5e28f46acd327d874d9a51f0cd51e09af04caa8e8b596f25dd9fae261f8c21","abstract_canon_sha256":"0f0f836f313bd1733d09be86fb6af887731d2aef7c4dacbee15c03d1d06077d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:49.007936Z","signature_b64":"kILR4GZUEsmm7Vqad5k7Yj7JbdSw3gsphxz0S8hQfGcP2NxPp75qPdpEjUTDOIn1ZrLtj6Cy9hQx++KMNgeSAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f9c691711ad003dd688f67120cbde388f4c74a3e0d576df3b897a4834d88568","last_reissued_at":"2026-07-05T11:25:49.007484Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:49.007484Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SlimMoE: Structured Compression of Large MoE Models via Expert Slimming and Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chen Liang, Ilgee Hong, Tuo Zhao, Weizhu Chen, Young Jin Kim, Zichong Li, Zixuan Zhang","submitted_at":"2025-06-23T07:15:59Z","abstract_excerpt":"The Mixture of Experts (MoE) architecture has emerged as a powerful paradigm for scaling large language models (LLMs) while maintaining inference efficiency. However, their enormous memory requirements make them prohibitively expensive to fine-tune or deploy in resource-constrained environments. To address this challenge, we introduce SlimMoE, a multi-stage compression framework for transforming large MoE models into much smaller, efficient variants without incurring the prohibitive costs of training from scratch. Our method systematically reduces parameter counts by slimming experts and trans"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.18349","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.18349/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.18349","created_at":"2026-07-05T11:25:49.007556+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.18349v1","created_at":"2026-07-05T11:25:49.007556+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.18349","created_at":"2026-07-05T11:25:49.007556+00:00"},{"alias_kind":"pith_short_12","alias_value":"J6OGSFYRVUAD","created_at":"2026-07-05T11:25:49.007556+00:00"},{"alias_kind":"pith_short_16","alias_value":"J6OGSFYRVUAD3VUI","created_at":"2026-07-05T11:25:49.007556+00:00"},{"alias_kind":"pith_short_8","alias_value":"J6OGSFYR","created_at":"2026-07-05T11:25:49.007556+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28207","citing_title":"Pruning and Distilling Mixture-of-Experts into Dense Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08738","citing_title":"SlimQwen: Exploring the Pruning and Distillation in Large MoE Model Pre-training","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08738","citing_title":"SlimQwen: Exploring the Pruning and Distillation in Large MoE Model Pre-training","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC","json":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC.json","graph_json":"https://pith.science/api/pith-number/J6OGSFYRVUAD3VUI6ZYSBS66HC/graph.json","events_json":"https://pith.science/api/pith-number/J6OGSFYRVUAD3VUI6ZYSBS66HC/events.json","paper":"https://pith.science/paper/J6OGSFYR"},"agent_actions":{"view_html":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC","download_json":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC.json","view_paper":"https://pith.science/paper/J6OGSFYR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.18349&json=true","fetch_graph":"https://pith.science/api/pith-number/J6OGSFYRVUAD3VUI6ZYSBS66HC/graph.json","fetch_events":"https://pith.science/api/pith-number/J6OGSFYRVUAD3VUI6ZYSBS66HC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC/action/storage_attestation","attest_author":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC/action/author_attestation","sign_citation":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC/action/citation_signature","submit_replication":"https://pith.science/pith/J6OGSFYRVUAD3VUI6ZYSBS66HC/action/replication_record"}},"created_at":"2026-07-05T11:25:49.007556+00:00","updated_at":"2026-07-05T11:25:49.007556+00:00"}