{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YAZCFPFMQDBEIH3UG4YH5A4CJB","short_pith_number":"pith:YAZCFPFM","schema_version":"1.0","canonical_sha256":"c03222bcac80c2441f7437307e8382485dddd493e106698c64e2807723d107e7","source":{"kind":"arxiv","id":"2407.00945","version":1},"attestation_state":"computed","paper":{"title":"Efficient Expert Pruning for Sparse Mixture-of-Experts Language Models: Enhancing Performance and Reducing Inference Costs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Enshu Liu, Guohao Dai, Huazhong Yang, Junyi Zhu, Matthew B. Blaschko, Shengen Yan, Xuefei Ning, Yu Wang, Zinan Lin","submitted_at":"2024-07-01T03:57:35Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has led to architectures with billions to trillions of parameters, posing significant deployment challenges due to their substantial demands on memory, processing power, and energy consumption. Sparse Mixture-of-Experts (SMoE) architectures have emerged as a solution, activating only a subset of parameters per token, thereby achieving faster inference while maintaining performance. However, SMoE models still face limitations in broader deployment due to their large parameter counts and significant GPU memory requirements. In this work, we i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00945","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-01T03:57:35Z","cross_cats_sorted":[],"title_canon_sha256":"126003e00ad1463443b6f4ccbb780165a469441e60709992a75bad0646adc5d7","abstract_canon_sha256":"5e6cc79f841159e1e9c1a50b675105cf88426efd92ef07244051f49d8ca2bd66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:26.613261Z","signature_b64":"YsRyolTtCx1p/34GIPXRin+S5qqldH8Xwxz2Qu9uCqUB+NIeRYaPGe8El3rm1EW09NjIOwXnNztRKVTDc2hsBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c03222bcac80c2441f7437307e8382485dddd493e106698c64e2807723d107e7","last_reissued_at":"2026-07-05T08:38:26.612918Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:26.612918Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Expert Pruning for Sparse Mixture-of-Experts Language Models: Enhancing Performance and Reducing Inference Costs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Enshu Liu, Guohao Dai, Huazhong Yang, Junyi Zhu, Matthew B. Blaschko, Shengen Yan, Xuefei Ning, Yu Wang, Zinan Lin","submitted_at":"2024-07-01T03:57:35Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has led to architectures with billions to trillions of parameters, posing significant deployment challenges due to their substantial demands on memory, processing power, and energy consumption. Sparse Mixture-of-Experts (SMoE) architectures have emerged as a solution, activating only a subset of parameters per token, thereby achieving faster inference while maintaining performance. However, SMoE models still face limitations in broader deployment due to their large parameter counts and significant GPU memory requirements. In this work, we i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00945","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00945/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00945","created_at":"2026-07-05T08:38:26.612971+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00945v1","created_at":"2026-07-05T08:38:26.612971+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00945","created_at":"2026-07-05T08:38:26.612971+00:00"},{"alias_kind":"pith_short_12","alias_value":"YAZCFPFMQDBE","created_at":"2026-07-05T08:38:26.612971+00:00"},{"alias_kind":"pith_short_16","alias_value":"YAZCFPFMQDBEIH3U","created_at":"2026-07-05T08:38:26.612971+00:00"},{"alias_kind":"pith_short_8","alias_value":"YAZCFPFM","created_at":"2026-07-05T08:38:26.612971+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05538","citing_title":"Less is MoE: Trimming Experts in Domain-Specialist Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18643","citing_title":"Post-Trained MoE Can Skip Half Experts via Self-Distillation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30876","citing_title":"dMoE: dLLMs with Learnable Block Experts","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18304","citing_title":"Attribution-Guided and Coverage-Maximized Pruning for Structural MoE Compression","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18643","citing_title":"Post-Trained MoE Can Skip Half Experts via Self-Distillation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06003","citing_title":"EvoESAP: Non-Uniform Expert Pruning for Sparse MoE","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13997","citing_title":"HodgeCover: Higher-Order Topological Coverage Drives Compression of Sparse Mixture-of-Experts","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06542","citing_title":"Does a Global Perspective Help Prune Sparse MoEs Elegantly?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20156","citing_title":"Temporally Extended Mixture-of-Experts Models","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB","json":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB.json","graph_json":"https://pith.science/api/pith-number/YAZCFPFMQDBEIH3UG4YH5A4CJB/graph.json","events_json":"https://pith.science/api/pith-number/YAZCFPFMQDBEIH3UG4YH5A4CJB/events.json","paper":"https://pith.science/paper/YAZCFPFM"},"agent_actions":{"view_html":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB","download_json":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB.json","view_paper":"https://pith.science/paper/YAZCFPFM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00945&json=true","fetch_graph":"https://pith.science/api/pith-number/YAZCFPFMQDBEIH3UG4YH5A4CJB/graph.json","fetch_events":"https://pith.science/api/pith-number/YAZCFPFMQDBEIH3UG4YH5A4CJB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB/action/storage_attestation","attest_author":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB/action/author_attestation","sign_citation":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB/action/citation_signature","submit_replication":"https://pith.science/pith/YAZCFPFMQDBEIH3UG4YH5A4CJB/action/replication_record"}},"created_at":"2026-07-05T08:38:26.612971+00:00","updated_at":"2026-07-05T08:38:26.612971+00:00"}