{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A6CUVYJCZCITLGDR5SCWGRWYOI","short_pith_number":"pith:A6CUVYJC","schema_version":"1.0","canonical_sha256":"07854ae122c891359871ec856346d8722b0245a65de3277434884f9eda4a1244","source":{"kind":"arxiv","id":"2406.18219","version":3},"attestation_state":"computed","paper":{"title":"A Closer Look into Mixture-of-Experts in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jie Fu, Ka Man Lo, Zeyu Huang, Zihan Qiu, Zili Wang","submitted_at":"2024-06-26T10:07:57Z","abstract_excerpt":"Mixture-of-experts (MoE) is gaining increasing attention due to its unique properties and remarkable performance, especially for language tasks. By sparsely activating a subset of parameters for each token, MoE architecture could increase the model size without sacrificing computational efficiency, achieving a better trade-off between performance and training costs. However, the underlying mechanism of MoE still lacks further exploration, and its modularization degree remains questionable. In this paper, we make an initial attempt to understand the inner workings of MoE-based large language mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18219","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-26T10:07:57Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9d28bd77131343298fcbec6dd04b0f0a33eb97e2b3ae5c17c9ab4d30aba0a6dd","abstract_canon_sha256":"0d9df0880d2e22b9377413c871c5cba5c40e85e823e19474708dbb7fc03a9dab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:02.089870Z","signature_b64":"WdiMHYJe8SyRcDPl+wTChQyHQSBJOmIg4L5fte6cUR4C6pHbOnjEnqK41L8Sm7AiEDy8IohkXrAcVJeqMOTvDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07854ae122c891359871ec856346d8722b0245a65de3277434884f9eda4a1244","last_reissued_at":"2026-07-05T11:25:02.089355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:02.089355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Closer Look into Mixture-of-Experts in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Jie Fu, Ka Man Lo, Zeyu Huang, Zihan Qiu, Zili Wang","submitted_at":"2024-06-26T10:07:57Z","abstract_excerpt":"Mixture-of-experts (MoE) is gaining increasing attention due to its unique properties and remarkable performance, especially for language tasks. By sparsely activating a subset of parameters for each token, MoE architecture could increase the model size without sacrificing computational efficiency, achieving a better trade-off between performance and training costs. However, the underlying mechanism of MoE still lacks further exploration, and its modularization degree remains questionable. In this paper, we make an initial attempt to understand the inner workings of MoE-based large language mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18219","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18219/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18219","created_at":"2026-07-05T11:25:02.089415+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18219v3","created_at":"2026-07-05T11:25:02.089415+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18219","created_at":"2026-07-05T11:25:02.089415+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6CUVYJCZCIT","created_at":"2026-07-05T11:25:02.089415+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6CUVYJCZCITLGDR","created_at":"2026-07-05T11:25:02.089415+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6CUVYJC","created_at":"2026-07-05T11:25:02.089415+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21645","citing_title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12119","citing_title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20610","citing_title":"Beyond Routing: Characterising Expert Tuning and Representation in Vision Mixture-of-Experts","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI","json":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI.json","graph_json":"https://pith.science/api/pith-number/A6CUVYJCZCITLGDR5SCWGRWYOI/graph.json","events_json":"https://pith.science/api/pith-number/A6CUVYJCZCITLGDR5SCWGRWYOI/events.json","paper":"https://pith.science/paper/A6CUVYJC"},"agent_actions":{"view_html":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI","download_json":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI.json","view_paper":"https://pith.science/paper/A6CUVYJC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18219&json=true","fetch_graph":"https://pith.science/api/pith-number/A6CUVYJCZCITLGDR5SCWGRWYOI/graph.json","fetch_events":"https://pith.science/api/pith-number/A6CUVYJCZCITLGDR5SCWGRWYOI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI/action/storage_attestation","attest_author":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI/action/author_attestation","sign_citation":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI/action/citation_signature","submit_replication":"https://pith.science/pith/A6CUVYJCZCITLGDR5SCWGRWYOI/action/replication_record"}},"created_at":"2026-07-05T11:25:02.089415+00:00","updated_at":"2026-07-05T11:25:02.089415+00:00"}