{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IK2WNORZX5NLUXFPAJYRIUL6HD","short_pith_number":"pith:IK2WNORZ","schema_version":"1.0","canonical_sha256":"42b566ba39bf5aba5caf027114517e38e2640a10275709dda77f00cbf43044e7","source":{"kind":"arxiv","id":"2405.16646","version":3},"attestation_state":"computed","paper":{"title":"A Provably Effective Method for Pruning Experts in Fine-tuned Sparse Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Christopher Carothers, Kaoutar El Maghraoui, Meng Wang, Mohammed Nowaz Rabbani Chowdhury, Naigang Wang, Pin-Yu Chen","submitted_at":"2024-05-26T17:52:58Z","abstract_excerpt":"The sparsely gated mixture of experts (MoE) architecture sends different inputs to different subnetworks, i.e., experts, through trainable routers. MoE reduces the training computation significantly for large models, but its deployment can be still memory or computation expensive for some downstream tasks. Model pruning is a popular approach to reduce inference computation, but its application in MoE architecture is largely unexplored. To the best of our knowledge, this paper provides the first provably efficient technique for pruning experts in finetuned MoE models. We theoretically prove tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16646","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-26T17:52:58Z","cross_cats_sorted":[],"title_canon_sha256":"44aecbc46c9831e9b8407065bcb03c3413291ed3dded18dde536d2bdd324962d","abstract_canon_sha256":"8669266a2a056ff308998f3914b95854a805209c62b554a3c4f138ca999efce9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:05.144949Z","signature_b64":"Zm8aWPH0UkAKKGsiEovPqqN0y3fGufyGhJuBE7al6v8e1Cg9qWyXZa97zKXDZBOArRNKOKcvhExDR1KtBjytAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42b566ba39bf5aba5caf027114517e38e2640a10275709dda77f00cbf43044e7","last_reissued_at":"2026-07-05T08:25:05.144463Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:05.144463Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Provably Effective Method for Pruning Experts in Fine-tuned Sparse Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Christopher Carothers, Kaoutar El Maghraoui, Meng Wang, Mohammed Nowaz Rabbani Chowdhury, Naigang Wang, Pin-Yu Chen","submitted_at":"2024-05-26T17:52:58Z","abstract_excerpt":"The sparsely gated mixture of experts (MoE) architecture sends different inputs to different subnetworks, i.e., experts, through trainable routers. MoE reduces the training computation significantly for large models, but its deployment can be still memory or computation expensive for some downstream tasks. Model pruning is a popular approach to reduce inference computation, but its application in MoE architecture is largely unexplored. To the best of our knowledge, this paper provides the first provably efficient technique for pruning experts in finetuned MoE models. We theoretically prove tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16646","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16646/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16646","created_at":"2026-07-05T08:25:05.144522+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16646v3","created_at":"2026-07-05T08:25:05.144522+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16646","created_at":"2026-07-05T08:25:05.144522+00:00"},{"alias_kind":"pith_short_12","alias_value":"IK2WNORZX5NL","created_at":"2026-07-05T08:25:05.144522+00:00"},{"alias_kind":"pith_short_16","alias_value":"IK2WNORZX5NLUXFP","created_at":"2026-07-05T08:25:05.144522+00:00"},{"alias_kind":"pith_short_8","alias_value":"IK2WNORZ","created_at":"2026-07-05T08:25:05.144522+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30876","citing_title":"dMoE: dLLMs with Learnable Block Experts","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18304","citing_title":"Attribution-Guided and Coverage-Maximized Pruning for Structural MoE Compression","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02715","citing_title":"FluxMoE: Decoupling Expert Residency for High-Performance MoE Serving","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06542","citing_title":"Does a Global Perspective Help Prune Sparse MoEs Elegantly?","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD","json":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD.json","graph_json":"https://pith.science/api/pith-number/IK2WNORZX5NLUXFPAJYRIUL6HD/graph.json","events_json":"https://pith.science/api/pith-number/IK2WNORZX5NLUXFPAJYRIUL6HD/events.json","paper":"https://pith.science/paper/IK2WNORZ"},"agent_actions":{"view_html":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD","download_json":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD.json","view_paper":"https://pith.science/paper/IK2WNORZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16646&json=true","fetch_graph":"https://pith.science/api/pith-number/IK2WNORZX5NLUXFPAJYRIUL6HD/graph.json","fetch_events":"https://pith.science/api/pith-number/IK2WNORZX5NLUXFPAJYRIUL6HD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD/action/storage_attestation","attest_author":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD/action/author_attestation","sign_citation":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD/action/citation_signature","submit_replication":"https://pith.science/pith/IK2WNORZX5NLUXFPAJYRIUL6HD/action/replication_record"}},"created_at":"2026-07-05T08:25:05.144522+00:00","updated_at":"2026-07-05T08:25:05.144522+00:00"}