{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6ZMBWR7RKHI5LIMXEQMBMI6ASF","short_pith_number":"pith:6ZMBWR7R","schema_version":"1.0","canonical_sha256":"f6581b47f151d1d5a19724181623c09152f1bf4392706dd521b3f76ce51a5fa4","source":{"kind":"arxiv","id":"2103.16716","version":1},"attestation_state":"computed","paper":{"title":"BASE Layers: Simplifying Training of Large, Sparse Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Luke Zettlemoyer, Mike Lewis, Naman Goyal, Shruti Bhosale, Tim Dettmers","submitted_at":"2021-03-30T23:08:32Z","abstract_excerpt":"We introduce a new balanced assignment of experts (BASE) layer for large language models that greatly simplifies existing high capacity sparse layers. Sparse layers can dramatically improve the efficiency of training and inference by routing each token to specialized expert modules that contain only a small fraction of the model parameters. However, it can be difficult to learn balanced routing functions that make full use of the available experts; existing approaches typically use routing heuristics or auxiliary expert-balancing loss functions. In contrast, we formulate token-to-expert alloca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.16716","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-03-30T23:08:32Z","cross_cats_sorted":[],"title_canon_sha256":"59d2210a06988bd5fbead419a19561b35385f6bcda7b12610039451d0c926f68","abstract_canon_sha256":"1435526101240ae00dcf3a90b5615e99c6e937ed96b7e79d31dd7d19a62e169b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:27:57.356328Z","signature_b64":"///2QFXV8So3JB6ZQc5wnUZu7GQdyKHUveMX7mxdkQXrOAM7cSnX/dEoJw6enFGx9WtOAgOvRj9/lO+AWd1DDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6581b47f151d1d5a19724181623c09152f1bf4392706dd521b3f76ce51a5fa4","last_reissued_at":"2026-07-05T02:27:57.355877Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:27:57.355877Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BASE Layers: Simplifying Training of Large, Sparse Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Luke Zettlemoyer, Mike Lewis, Naman Goyal, Shruti Bhosale, Tim Dettmers","submitted_at":"2021-03-30T23:08:32Z","abstract_excerpt":"We introduce a new balanced assignment of experts (BASE) layer for large language models that greatly simplifies existing high capacity sparse layers. Sparse layers can dramatically improve the efficiency of training and inference by routing each token to specialized expert modules that contain only a small fraction of the model parameters. However, it can be difficult to learn balanced routing functions that make full use of the available experts; existing approaches typically use routing heuristics or auxiliary expert-balancing loss functions. In contrast, we formulate token-to-expert alloca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.16716","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.16716/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.16716","created_at":"2026-07-05T02:27:57.355939+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.16716v1","created_at":"2026-07-05T02:27:57.355939+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.16716","created_at":"2026-07-05T02:27:57.355939+00:00"},{"alias_kind":"pith_short_12","alias_value":"6ZMBWR7RKHI5","created_at":"2026-07-05T02:27:57.355939+00:00"},{"alias_kind":"pith_short_16","alias_value":"6ZMBWR7RKHI5LIMX","created_at":"2026-07-05T02:27:57.355939+00:00"},{"alias_kind":"pith_short_8","alias_value":"6ZMBWR7R","created_at":"2026-07-05T02:27:57.355939+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29121","citing_title":"A Minimal Bifurcation Model of Load Imbalance in a Softmax Mixture-of-Experts Router","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2208.07339","citing_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2202.08906","citing_title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2205.01068","citing_title":"OPT: Open Pre-trained Transformer Language Models","ref_index":170,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF","json":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF.json","graph_json":"https://pith.science/api/pith-number/6ZMBWR7RKHI5LIMXEQMBMI6ASF/graph.json","events_json":"https://pith.science/api/pith-number/6ZMBWR7RKHI5LIMXEQMBMI6ASF/events.json","paper":"https://pith.science/paper/6ZMBWR7R"},"agent_actions":{"view_html":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF","download_json":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF.json","view_paper":"https://pith.science/paper/6ZMBWR7R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.16716&json=true","fetch_graph":"https://pith.science/api/pith-number/6ZMBWR7RKHI5LIMXEQMBMI6ASF/graph.json","fetch_events":"https://pith.science/api/pith-number/6ZMBWR7RKHI5LIMXEQMBMI6ASF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF/action/storage_attestation","attest_author":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF/action/author_attestation","sign_citation":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF/action/citation_signature","submit_replication":"https://pith.science/pith/6ZMBWR7RKHI5LIMXEQMBMI6ASF/action/replication_record"}},"created_at":"2026-07-05T02:27:57.355939+00:00","updated_at":"2026-07-05T02:27:57.355939+00:00"}