{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MEF5PYBHQ75GIMQR6LPO4XBCTN","short_pith_number":"pith:MEF5PYBH","schema_version":"1.0","canonical_sha256":"610bd7e02787fa643211f2deee5c229b4475ed6b16cb7d7f0d3bc58988e4c4e3","source":{"kind":"arxiv","id":"2411.15708","version":1},"attestation_state":"computed","paper":{"title":"LLaMA-MoE v2: Exploring Sparsity of LLaMA from Perspective of Mixture-of-Experts with Post-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daize Dong, Tong Zhu, Weigao Sun, Xiaoye Qu, Xuyang Hu, Yu Cheng","submitted_at":"2024-11-24T04:26:04Z","abstract_excerpt":"Recently, inspired by the concept of sparsity, Mixture-of-Experts (MoE) models have gained increasing popularity for scaling model size while keeping the number of activated parameters constant. In this study, we thoroughly investigate the sparsity of the dense LLaMA model by constructing MoE for both the attention (i.e., Attention MoE) and MLP (i.e., MLP MoE) modules in the transformer blocks. Specifically, we investigate different expert construction methods and granularities under the same activation conditions to analyze the impact of sparsifying the model. Additionally, to comprehensively"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15708","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-24T04:26:04Z","cross_cats_sorted":[],"title_canon_sha256":"c5272a1e31a7ab48c7a6cb9a870bce3e4f6fb01895b279a6034dde731a65d981","abstract_canon_sha256":"eadc5054343002e63027c2d0a40e8999a671d33dc9c8207995a91b996ba8c51d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:39.700804Z","signature_b64":"QqP/KereAPvnkjvc077gZp1Qy3RLkHZOU/2TiN9kCDkLZRMJEDrWNHHRG3suyiTjtd7y36bqSO2YSCj09ceCBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"610bd7e02787fa643211f2deee5c229b4475ed6b16cb7d7f0d3bc58988e4c4e3","last_reissued_at":"2026-07-05T09:39:39.700267Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:39.700267Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLaMA-MoE v2: Exploring Sparsity of LLaMA from Perspective of Mixture-of-Experts with Post-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daize Dong, Tong Zhu, Weigao Sun, Xiaoye Qu, Xuyang Hu, Yu Cheng","submitted_at":"2024-11-24T04:26:04Z","abstract_excerpt":"Recently, inspired by the concept of sparsity, Mixture-of-Experts (MoE) models have gained increasing popularity for scaling model size while keeping the number of activated parameters constant. In this study, we thoroughly investigate the sparsity of the dense LLaMA model by constructing MoE for both the attention (i.e., Attention MoE) and MLP (i.e., MLP MoE) modules in the transformer blocks. Specifically, we investigate different expert construction methods and granularities under the same activation conditions to analyze the impact of sparsifying the model. Additionally, to comprehensively"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15708","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15708/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15708","created_at":"2026-07-05T09:39:39.700341+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15708v1","created_at":"2026-07-05T09:39:39.700341+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15708","created_at":"2026-07-05T09:39:39.700341+00:00"},{"alias_kind":"pith_short_12","alias_value":"MEF5PYBHQ75G","created_at":"2026-07-05T09:39:39.700341+00:00"},{"alias_kind":"pith_short_16","alias_value":"MEF5PYBHQ75GIMQR","created_at":"2026-07-05T09:39:39.700341+00:00"},{"alias_kind":"pith_short_8","alias_value":"MEF5PYBH","created_at":"2026-07-05T09:39:39.700341+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20945","citing_title":"Grouped Query Experts: Mixture-of-Experts on GQA Self-Attention","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24681","citing_title":"Mix-MoE: Improving Multilingual Machine Translation of Large Language Models through Mixed MoEs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04416","citing_title":"Analytical FFN-to-MoE Restructuring via Activation Pattern Analysis","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05225","citing_title":"MACS: Modality-Aware Capacity Scaling for Efficient Multimodal MoE Inference","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN","json":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN.json","graph_json":"https://pith.science/api/pith-number/MEF5PYBHQ75GIMQR6LPO4XBCTN/graph.json","events_json":"https://pith.science/api/pith-number/MEF5PYBHQ75GIMQR6LPO4XBCTN/events.json","paper":"https://pith.science/paper/MEF5PYBH"},"agent_actions":{"view_html":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN","download_json":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN.json","view_paper":"https://pith.science/paper/MEF5PYBH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15708&json=true","fetch_graph":"https://pith.science/api/pith-number/MEF5PYBHQ75GIMQR6LPO4XBCTN/graph.json","fetch_events":"https://pith.science/api/pith-number/MEF5PYBHQ75GIMQR6LPO4XBCTN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN/action/storage_attestation","attest_author":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN/action/author_attestation","sign_citation":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN/action/citation_signature","submit_replication":"https://pith.science/pith/MEF5PYBHQ75GIMQR6LPO4XBCTN/action/replication_record"}},"created_at":"2026-07-05T09:39:39.700341+00:00","updated_at":"2026-07-05T09:39:39.700341+00:00"}