{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FNFL72LQACOSTFXF3IMGYSU35H","short_pith_number":"pith:FNFL72LQ","schema_version":"1.0","canonical_sha256":"2b4abfe970009d2996e5da186c4a9be9c00ea182563770ea97574e77038dfe53","source":{"kind":"arxiv","id":"2410.12013","version":1},"attestation_state":"computed","paper":{"title":"MoE-Pruner: Pruning Mixture-of-Experts Large Language Model using the Hints from Its Router","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"An Xu, Cong Xie, Ding Zhou, Xin Liu, Xue Lin, Yanyue Xie, Yanzhi Wang, Zhi Zhang, Ziang Song","submitted_at":"2024-10-15T19:22:27Z","abstract_excerpt":"Mixture-of-Experts (MoE) architectures face challenges such as high memory consumption and redundancy in experts. Pruning MoE can reduce network weights while maintaining model performance. Motivated by the recent observation of emergent large magnitude features in Large Language Models (LLM) and MoE routing policy, we propose MoE-Pruner, a method that prunes weights with the smallest magnitudes multiplied by the corresponding input activations and router weights, on each output neuron. Our pruning method is one-shot, requiring no retraining or weight updates. We evaluate our method on Mixtral"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12013","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-15T19:22:27Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"98c63912ea6f604dce08026be7502169c5d2e63e9611cad11ef6488ed14b1c9a","abstract_canon_sha256":"820c0f9a12d76d6f079035131b3fe84f5275954d9b7cb4520c17f99e7128b552"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:12.495687Z","signature_b64":"6/oq4/8RkJSwSHXk++y9JeFNDsylLuedTurWqjq0cz0atL2oQbZ6JeW8eukhPDGtXZpSBELZ/TS/ZTfefcuBCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b4abfe970009d2996e5da186c4a9be9c00ea182563770ea97574e77038dfe53","last_reissued_at":"2026-07-05T09:21:12.495191Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:12.495191Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoE-Pruner: Pruning Mixture-of-Experts Large Language Model using the Hints from Its Router","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"An Xu, Cong Xie, Ding Zhou, Xin Liu, Xue Lin, Yanyue Xie, Yanzhi Wang, Zhi Zhang, Ziang Song","submitted_at":"2024-10-15T19:22:27Z","abstract_excerpt":"Mixture-of-Experts (MoE) architectures face challenges such as high memory consumption and redundancy in experts. Pruning MoE can reduce network weights while maintaining model performance. Motivated by the recent observation of emergent large magnitude features in Large Language Models (LLM) and MoE routing policy, we propose MoE-Pruner, a method that prunes weights with the smallest magnitudes multiplied by the corresponding input activations and router weights, on each output neuron. Our pruning method is one-shot, requiring no retraining or weight updates. We evaluate our method on Mixtral"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12013","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12013","created_at":"2026-07-05T09:21:12.495250+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12013v1","created_at":"2026-07-05T09:21:12.495250+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12013","created_at":"2026-07-05T09:21:12.495250+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNFL72LQACOS","created_at":"2026-07-05T09:21:12.495250+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNFL72LQACOSTFXF","created_at":"2026-07-05T09:21:12.495250+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNFL72LQ","created_at":"2026-07-05T09:21:12.495250+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05538","citing_title":"Less is MoE: Trimming Experts in Domain-Specialist Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29982","citing_title":"Beyond Uniform Experts: Cost-Aware Expert Execution for Efficient Multi-Device MoE Inference","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28207","citing_title":"Pruning and Distilling Mixture-of-Experts into Dense Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06003","citing_title":"EvoESAP: Non-Uniform Expert Pruning for Sparse MoE","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13997","citing_title":"HodgeCover: Higher-Order Topological Coverage Drives Compression of Sparse Mixture-of-Experts","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20156","citing_title":"Temporally Extended Mixture-of-Experts Models","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H","json":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H.json","graph_json":"https://pith.science/api/pith-number/FNFL72LQACOSTFXF3IMGYSU35H/graph.json","events_json":"https://pith.science/api/pith-number/FNFL72LQACOSTFXF3IMGYSU35H/events.json","paper":"https://pith.science/paper/FNFL72LQ"},"agent_actions":{"view_html":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H","download_json":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H.json","view_paper":"https://pith.science/paper/FNFL72LQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12013&json=true","fetch_graph":"https://pith.science/api/pith-number/FNFL72LQACOSTFXF3IMGYSU35H/graph.json","fetch_events":"https://pith.science/api/pith-number/FNFL72LQACOSTFXF3IMGYSU35H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H/action/storage_attestation","attest_author":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H/action/author_attestation","sign_citation":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H/action/citation_signature","submit_replication":"https://pith.science/pith/FNFL72LQACOSTFXF3IMGYSU35H/action/replication_record"}},"created_at":"2026-07-05T09:21:12.495250+00:00","updated_at":"2026-07-05T09:21:12.495250+00:00"}