{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SZKVNTAPGSB7AXNWLOYSALUMMI","short_pith_number":"pith:SZKVNTAP","schema_version":"1.0","canonical_sha256":"965556cc0f3483f05db65bb1202e8c620e7ca9b8ff98d37200ab58b5f0d1ff42","source":{"kind":"arxiv","id":"2110.01786","version":3},"attestation_state":"computed","paper":{"title":"MoEfication: Transformer Feed-forward Layers are Mixtures of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jie Zhou, Maosong Sun, Peng Li, Yankai Lin, Zhengyan Zhang, Zhiyuan Liu","submitted_at":"2021-10-05T02:14:38Z","abstract_excerpt":"Recent work has shown that feed-forward networks (FFNs) in pre-trained Transformers are a key component, storing various linguistic and factual knowledge. However, the computational patterns of FFNs are still unclear. In this work, we study the computational patterns of FFNs and observe that most inputs only activate a tiny ratio of neurons of FFNs. This phenomenon is similar to the sparsity of the human brain, which drives research on functional partitions of the human brain. To verify whether functional partitions also emerge in FFNs, we propose to convert a model into its MoE version with t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.01786","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-10-05T02:14:38Z","cross_cats_sorted":[],"title_canon_sha256":"e1f2dbc7dd5bb46254269a8dbbb216be63e83b892cf5214d0ce9753b964fdf6f","abstract_canon_sha256":"3b9d41d619d6a1fb994625259c99300dbdd0cd31177d03c29016fe55d04c2114"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:11:27.414049Z","signature_b64":"C1rlfjZUi3zAvJy+ixeYxSENUdI8C7aVMdwUacxqzUsdMg/pOk8eKHUAJnipggtgeNaR/Whj/bkt6tB/x3a6Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"965556cc0f3483f05db65bb1202e8c620e7ca9b8ff98d37200ab58b5f0d1ff42","last_reissued_at":"2026-07-05T04:11:27.413590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:11:27.413590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoEfication: Transformer Feed-forward Layers are Mixtures of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jie Zhou, Maosong Sun, Peng Li, Yankai Lin, Zhengyan Zhang, Zhiyuan Liu","submitted_at":"2021-10-05T02:14:38Z","abstract_excerpt":"Recent work has shown that feed-forward networks (FFNs) in pre-trained Transformers are a key component, storing various linguistic and factual knowledge. However, the computational patterns of FFNs are still unclear. In this work, we study the computational patterns of FFNs and observe that most inputs only activate a tiny ratio of neurons of FFNs. This phenomenon is similar to the sparsity of the human brain, which drives research on functional partitions of the human brain. To verify whether functional partitions also emerge in FFNs, we propose to convert a model into its MoE version with t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.01786","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.01786/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.01786","created_at":"2026-07-05T04:11:27.413650+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.01786v3","created_at":"2026-07-05T04:11:27.413650+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.01786","created_at":"2026-07-05T04:11:27.413650+00:00"},{"alias_kind":"pith_short_12","alias_value":"SZKVNTAPGSB7","created_at":"2026-07-05T04:11:27.413650+00:00"},{"alias_kind":"pith_short_16","alias_value":"SZKVNTAPGSB7AXNW","created_at":"2026-07-05T04:11:27.413650+00:00"},{"alias_kind":"pith_short_8","alias_value":"SZKVNTAP","created_at":"2026-07-05T04:11:27.413650+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08814","citing_title":"STAR: Rethinking MoE Routing as Structure-Aware Subspace Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04416","citing_title":"Analytical FFN-to-MoE Restructuring via Activation Pattern Analysis","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12119","citing_title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI","json":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI.json","graph_json":"https://pith.science/api/pith-number/SZKVNTAPGSB7AXNWLOYSALUMMI/graph.json","events_json":"https://pith.science/api/pith-number/SZKVNTAPGSB7AXNWLOYSALUMMI/events.json","paper":"https://pith.science/paper/SZKVNTAP"},"agent_actions":{"view_html":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI","download_json":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI.json","view_paper":"https://pith.science/paper/SZKVNTAP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.01786&json=true","fetch_graph":"https://pith.science/api/pith-number/SZKVNTAPGSB7AXNWLOYSALUMMI/graph.json","fetch_events":"https://pith.science/api/pith-number/SZKVNTAPGSB7AXNWLOYSALUMMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI/action/storage_attestation","attest_author":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI/action/author_attestation","sign_citation":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI/action/citation_signature","submit_replication":"https://pith.science/pith/SZKVNTAPGSB7AXNWLOYSALUMMI/action/replication_record"}},"created_at":"2026-07-05T04:11:27.413650+00:00","updated_at":"2026-07-05T04:11:27.413650+00:00"}