{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MRLI72XKPNEHZLBBCLOP7UHJJL","short_pith_number":"pith:MRLI72XK","schema_version":"1.0","canonical_sha256":"64568feaea7b487cac2112dcffd0e94af631ed27c3085124271fd65ba4431259","source":{"kind":"arxiv","id":"2401.04081","version":2},"attestation_state":"computed","paper":{"title":"MoE-Mamba: Efficient Selective State Space Models with Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jakub Krajewski, Jan Ludziejewski, Kamil Ciebiera, Krystian Kr\\'ol, Maciej Pi\\'oro, Marek Cygan, Micha{\\l} Krutul, Piotr Mi{\\l}o\\'s, Sebastian Jaszczur, Szymon Antoniak","submitted_at":"2024-01-08T18:35:07Z","abstract_excerpt":"State Space Models (SSMs) have become serious contenders in the field of sequential modeling, challenging the dominance of Transformers. At the same time, Mixture of Experts (MoE) has significantly improved Transformer-based Large Language Models, including recent state-of-the-art open models. We propose that to unlock the potential of SSMs for scaling, they should be combined with MoE. We showcase this on Mamba, a recent SSM-based model that achieves remarkable performance. Our model, MoE-Mamba, outperforms both Mamba and baseline Transformer-MoE. In particular, MoE-Mamba reaches the same per"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.04081","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-01-08T18:35:07Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"dade418c5d8c36318f54866fbb4ef004ed02d988699e93352c20ada915178f7f","abstract_canon_sha256":"cd2157d5fc00ed8a8a2216d8d0018a0243d31901148cce02fbc86caf5976f8c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:17.173204Z","signature_b64":"nfnrJHP3RdWZad+jXXS4lRJ9CnKFfLKftyXQVfVYuf4qco8u1etmJPHBu+uUGGPX6IN8dcZ0n+yynGnk//CaCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64568feaea7b487cac2112dcffd0e94af631ed27c3085124271fd65ba4431259","last_reissued_at":"2026-07-05T07:49:17.172720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:17.172720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MoE-Mamba: Efficient Selective State Space Models with Mixture of Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jakub Krajewski, Jan Ludziejewski, Kamil Ciebiera, Krystian Kr\\'ol, Maciej Pi\\'oro, Marek Cygan, Micha{\\l} Krutul, Piotr Mi{\\l}o\\'s, Sebastian Jaszczur, Szymon Antoniak","submitted_at":"2024-01-08T18:35:07Z","abstract_excerpt":"State Space Models (SSMs) have become serious contenders in the field of sequential modeling, challenging the dominance of Transformers. At the same time, Mixture of Experts (MoE) has significantly improved Transformer-based Large Language Models, including recent state-of-the-art open models. We propose that to unlock the potential of SSMs for scaling, they should be combined with MoE. We showcase this on Mamba, a recent SSM-based model that achieves remarkable performance. Our model, MoE-Mamba, outperforms both Mamba and baseline Transformer-MoE. In particular, MoE-Mamba reaches the same per"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.04081","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.04081/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.04081","created_at":"2026-07-05T07:49:17.172776+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.04081v2","created_at":"2026-07-05T07:49:17.172776+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.04081","created_at":"2026-07-05T07:49:17.172776+00:00"},{"alias_kind":"pith_short_12","alias_value":"MRLI72XKPNEH","created_at":"2026-07-05T07:49:17.172776+00:00"},{"alias_kind":"pith_short_16","alias_value":"MRLI72XKPNEHZLBB","created_at":"2026-07-05T07:49:17.172776+00:00"},{"alias_kind":"pith_short_8","alias_value":"MRLI72XK","created_at":"2026-07-05T07:49:17.172776+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10395","citing_title":"Efficient RWKV-based Representation Learning for 3D Point Clouds","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2404.07106","citing_title":"3DMambaComplete: Exploring Structured State Space Model for Point Cloud Completion","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2403.19887","citing_title":"Jamba: A Hybrid Transformer-Mamba Language Model","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12257","citing_title":"Style-Decoupled Adaptive Routing Network for Underwater Image Enhancement","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL","json":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL.json","graph_json":"https://pith.science/api/pith-number/MRLI72XKPNEHZLBBCLOP7UHJJL/graph.json","events_json":"https://pith.science/api/pith-number/MRLI72XKPNEHZLBBCLOP7UHJJL/events.json","paper":"https://pith.science/paper/MRLI72XK"},"agent_actions":{"view_html":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL","download_json":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL.json","view_paper":"https://pith.science/paper/MRLI72XK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.04081&json=true","fetch_graph":"https://pith.science/api/pith-number/MRLI72XKPNEHZLBBCLOP7UHJJL/graph.json","fetch_events":"https://pith.science/api/pith-number/MRLI72XKPNEHZLBBCLOP7UHJJL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL/action/storage_attestation","attest_author":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL/action/author_attestation","sign_citation":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL/action/citation_signature","submit_replication":"https://pith.science/pith/MRLI72XKPNEHZLBBCLOP7UHJJL/action/replication_record"}},"created_at":"2026-07-05T07:49:17.172776+00:00","updated_at":"2026-07-05T07:49:17.172776+00:00"}