{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:AWAZZCU2OP44ZHBNYVNPGOKUMH","short_pith_number":"pith:AWAZZCU2","schema_version":"1.0","canonical_sha256":"05819c8a9a73f9cc9c2dc55af3395461f102622132a80b6963fc411a4d12629d","source":{"kind":"arxiv","id":"2209.01667","version":1},"attestation_state":"computed","paper":{"title":"A Review of Sparse Expert Models in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Barret Zoph, Jeff Dean, William Fedus","submitted_at":"2022-09-04T18:00:29Z","abstract_excerpt":"Sparse expert models are a thirty-year old concept re-emerging as a popular architecture in deep learning. This class of architecture encompasses Mixture-of-Experts, Switch Transformers, Routing Networks, BASE layers, and others, all with the unifying idea that each example is acted on by a subset of the parameters. By doing so, the degree of sparsity decouples the parameter count from the compute per example allowing for extremely large, but efficient models. The resulting models have demonstrated significant improvements across diverse domains such as natural language processing, computer vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.01667","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-09-04T18:00:29Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a31a80bc2365248ba1a0a196f16373999260e1d1fbcc2d084fc2f45a12c04c02","abstract_canon_sha256":"de53585c0d9c575f80c9755e7a536a800a01598ce0789fd92365f70730e56234"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:54:31.988604Z","signature_b64":"rNMKOtRdTmDJY364ARFjgvSiWo2m5Uocz0Ne/DRlQI8thM8EFTpgVkkmk3wH15uxD5SVyaXlu2y3TbEkIsXhCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"05819c8a9a73f9cc9c2dc55af3395461f102622132a80b6963fc411a4d12629d","last_reissued_at":"2026-07-05T04:54:31.988254Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:54:31.988254Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Review of Sparse Expert Models in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Barret Zoph, Jeff Dean, William Fedus","submitted_at":"2022-09-04T18:00:29Z","abstract_excerpt":"Sparse expert models are a thirty-year old concept re-emerging as a popular architecture in deep learning. This class of architecture encompasses Mixture-of-Experts, Switch Transformers, Routing Networks, BASE layers, and others, all with the unifying idea that each example is acted on by a subset of the parameters. By doing so, the degree of sparsity decouples the parameter count from the compute per example allowing for extremely large, but efficient models. The resulting models have demonstrated significant improvements across diverse domains such as natural language processing, computer vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.01667","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.01667/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.01667","created_at":"2026-07-05T04:54:31.988310+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.01667v1","created_at":"2026-07-05T04:54:31.988310+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.01667","created_at":"2026-07-05T04:54:31.988310+00:00"},{"alias_kind":"pith_short_12","alias_value":"AWAZZCU2OP44","created_at":"2026-07-05T04:54:31.988310+00:00"},{"alias_kind":"pith_short_16","alias_value":"AWAZZCU2OP44ZHBN","created_at":"2026-07-05T04:54:31.988310+00:00"},{"alias_kind":"pith_short_8","alias_value":"AWAZZCU2","created_at":"2026-07-05T04:54:31.988310+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10454","citing_title":"Entropy-Aware Domain-Routed Mixture-of-Experts Speech-LLM Framework: A Case Study of Multi-Domain Child-Adult ASR","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10431","citing_title":"Vision-Assisted Foundation Model for Solving Multi-Task Vehicle Routing Problems","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24270","citing_title":"Safety-Oriented Routing Analysis of Mixtral MoE Under Benign and Harmful Prompts","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2401.04088","citing_title":"Mixtral of Experts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17138","citing_title":"RAP: Runtime Adaptive Pruning for LLM Inference","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20837","citing_title":"ArchSIBench: Benchmarking the Architectural Spatial Intelligence of Vision-Language Models","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20414","citing_title":"RetroMotion: Retrocausal Motion Forecasting Models are Instructable","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2302.12288","citing_title":"ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11537","citing_title":"Fast MoE Inference via Predictive Prefetching and Expert Replication","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2209.10652","citing_title":"Toy Models of Superposition","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH","json":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH.json","graph_json":"https://pith.science/api/pith-number/AWAZZCU2OP44ZHBNYVNPGOKUMH/graph.json","events_json":"https://pith.science/api/pith-number/AWAZZCU2OP44ZHBNYVNPGOKUMH/events.json","paper":"https://pith.science/paper/AWAZZCU2"},"agent_actions":{"view_html":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH","download_json":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH.json","view_paper":"https://pith.science/paper/AWAZZCU2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.01667&json=true","fetch_graph":"https://pith.science/api/pith-number/AWAZZCU2OP44ZHBNYVNPGOKUMH/graph.json","fetch_events":"https://pith.science/api/pith-number/AWAZZCU2OP44ZHBNYVNPGOKUMH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH/action/storage_attestation","attest_author":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH/action/author_attestation","sign_citation":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH/action/citation_signature","submit_replication":"https://pith.science/pith/AWAZZCU2OP44ZHBNYVNPGOKUMH/action/replication_record"}},"created_at":"2026-07-05T04:54:31.988310+00:00","updated_at":"2026-07-05T04:54:31.988310+00:00"}