{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O22F5HK3KVJIVFBKB2XNW4U776","short_pith_number":"pith:O22F5HK3","schema_version":"1.0","canonical_sha256":"76b45e9d5b55528a942a0eaedb729fff8ce0bab57e8a8ab9ae7a8a4ad164e9de","source":{"kind":"arxiv","id":"2402.14800","version":2},"attestation_state":"computed","paper":{"title":"Not All Experts are Equal: Efficient Expert Pruning and Skipping for Mixture-of-Experts Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aojun Zhou, Bo Zhang, Hongsheng Li, Junchi Yan, Qi Liu, Siyuan Huang, Xudong Lu, Yuhui Xu","submitted_at":"2024-02-22T18:56:07Z","abstract_excerpt":"A pivotal advancement in the progress of large language models (LLMs) is the emergence of the Mixture-of-Experts (MoE) LLMs. Compared to traditional LLMs, MoE LLMs can achieve higher performance with fewer parameters, but it is still hard to deploy them due to their immense parameter sizes. Different from previous weight pruning methods that rely on specifically designed hardware, this paper mainly aims to enhance the deployment efficiency of MoE LLMs by introducing plug-and-play expert-level sparsification techniques. Specifically, we propose, for the first time to our best knowledge, post-tr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14800","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-22T18:56:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"feb735e9fd7e62b2883c24dce648267d9c1a4bc68fd77346e9f6a62cccbef930","abstract_canon_sha256":"37de7c1b5605434722e80a9cde42ad824871563555820b545086abb6456e7cc2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:00.947941Z","signature_b64":"j50/IvYYGutea2N8RMZf324J5NK27wCtiICTcBdtIn4qe74f5mnMhmG+o42hJE8dVnM1IU4uHb1OKYaoqpdeBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76b45e9d5b55528a942a0eaedb729fff8ce0bab57e8a8ab9ae7a8a4ad164e9de","last_reissued_at":"2026-07-05T08:25:00.947460Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:00.947460Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All Experts are Equal: Efficient Expert Pruning and Skipping for Mixture-of-Experts Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aojun Zhou, Bo Zhang, Hongsheng Li, Junchi Yan, Qi Liu, Siyuan Huang, Xudong Lu, Yuhui Xu","submitted_at":"2024-02-22T18:56:07Z","abstract_excerpt":"A pivotal advancement in the progress of large language models (LLMs) is the emergence of the Mixture-of-Experts (MoE) LLMs. Compared to traditional LLMs, MoE LLMs can achieve higher performance with fewer parameters, but it is still hard to deploy them due to their immense parameter sizes. Different from previous weight pruning methods that rely on specifically designed hardware, this paper mainly aims to enhance the deployment efficiency of MoE LLMs by introducing plug-and-play expert-level sparsification techniques. Specifically, we propose, for the first time to our best knowledge, post-tr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14800","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14800/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14800","created_at":"2026-07-05T08:25:00.947522+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14800v2","created_at":"2026-07-05T08:25:00.947522+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14800","created_at":"2026-07-05T08:25:00.947522+00:00"},{"alias_kind":"pith_short_12","alias_value":"O22F5HK3KVJI","created_at":"2026-07-05T08:25:00.947522+00:00"},{"alias_kind":"pith_short_16","alias_value":"O22F5HK3KVJIVFBK","created_at":"2026-07-05T08:25:00.947522+00:00"},{"alias_kind":"pith_short_8","alias_value":"O22F5HK3","created_at":"2026-07-05T08:25:00.947522+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18304","citing_title":"Attribution-Guided and Coverage-Maximized Pruning for Structural MoE Compression","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04160","citing_title":"Expert-Aware Refusal Steering","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29982","citing_title":"Beyond Uniform Experts: Cost-Aware Expert Execution for Efficient Multi-Device MoE Inference","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00079","citing_title":"BitsMoE: Efficient Spectral Energy-Guided Bit Allocation for MoE LLM Quantization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23078","citing_title":"GEMQ: Global Expert-Level Mixed-Precision Quantization for MoE LLMs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2412.00069","citing_title":"Condense, Don't Just Prune: Enhancing Efficiency and Performance in MoE Layer Pruning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2411.08982","citing_title":"Lynx: Enabling Efficient MoE Inference through Dynamic Batch-Aware Expert Selection","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04416","citing_title":"Analytical FFN-to-MoE Restructuring via Activation Pattern Analysis","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08738","citing_title":"SlimQwen: Exploring the Pruning and Distillation in Large MoE Model Pre-training","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15484","citing_title":"When Does Sparse MoE Help in Vision? The Role of Backbone Compute Leverage in Sparse Routing","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23638","citing_title":"LayerScope: Predictive Cross-Layer Scheduling for Efficient Multi-Batch MoE Inference on Legacy Servers","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":199,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06003","citing_title":"EvoESAP: Non-Uniform Expert Pruning for Sparse MoE","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":183,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02715","citing_title":"FluxMoE: Decoupling Expert Residency for High-Performance MoE Serving","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08738","citing_title":"SlimQwen: Exploring the Pruning and Distillation in Large MoE Model Pre-training","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23036","citing_title":"Preserving Long-Tailed Expert Information in Mixture-of-Experts Tuning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19835","citing_title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20156","citing_title":"Temporally Extended Mixture-of-Experts Models","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776","json":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776.json","graph_json":"https://pith.science/api/pith-number/O22F5HK3KVJIVFBKB2XNW4U776/graph.json","events_json":"https://pith.science/api/pith-number/O22F5HK3KVJIVFBKB2XNW4U776/events.json","paper":"https://pith.science/paper/O22F5HK3"},"agent_actions":{"view_html":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776","download_json":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776.json","view_paper":"https://pith.science/paper/O22F5HK3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14800&json=true","fetch_graph":"https://pith.science/api/pith-number/O22F5HK3KVJIVFBKB2XNW4U776/graph.json","fetch_events":"https://pith.science/api/pith-number/O22F5HK3KVJIVFBKB2XNW4U776/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776/action/storage_attestation","attest_author":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776/action/author_attestation","sign_citation":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776/action/citation_signature","submit_replication":"https://pith.science/pith/O22F5HK3KVJIVFBKB2XNW4U776/action/replication_record"}},"created_at":"2026-07-05T08:25:00.947522+00:00","updated_at":"2026-07-05T08:25:00.947522+00:00"}