{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WN2SAGEIELIWLF2ORF5HO2EY43","short_pith_number":"pith:WN2SAGEI","schema_version":"1.0","canonical_sha256":"b37520188822d165974e897a776898e6fa7860928404e772c9fb128f7df4add9","source":{"kind":"arxiv","id":"2309.04354","version":1},"attestation_state":"computed","paper":{"title":"Mobile V-MoEs: Scaling Down Vision Transformers via Sparse Mixture-of-Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Alexander Toshev, Bowen Zhang, Erik Daxberger, Floris Weers, Marcin Eichner, Michael Emmersberger, Ruoming Pang, Tom Gunter, Xianzhi Du, Yinfei Yang","submitted_at":"2023-09-08T14:24:10Z","abstract_excerpt":"Sparse Mixture-of-Experts models (MoEs) have recently gained popularity due to their ability to decouple model size from inference efficiency by only activating a small subset of the model parameters for any given input token. As such, sparse MoEs have enabled unprecedented scalability, resulting in tremendous successes across domains such as natural language processing and computer vision. In this work, we instead explore the use of sparse MoEs to scale-down Vision Transformers (ViTs) to make them more attractive for resource-constrained vision applications. To this end, we propose a simplifi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.04354","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-08T14:24:10Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"87b747e5d0b0cb68918b2f74d31def85a8042d67abceef4f3cbdd1429e89df3f","abstract_canon_sha256":"b9c4ba9c061089dc188243bc9b938947324c7f92c706394ffa3fb7d1b82ea2df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:58.691518Z","signature_b64":"GF5zfEDa3nNW9ns4AjdWTj/raCNB9QOyNVz0bFkH23GLtdtKaJ4zGS9cSXf5I8RP/26t0q6v0k7wL/E0Sx2QAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b37520188822d165974e897a776898e6fa7860928404e772c9fb128f7df4add9","last_reissued_at":"2026-07-05T06:48:58.690956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:58.690956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mobile V-MoEs: Scaling Down Vision Transformers via Sparse Mixture-of-Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Alexander Toshev, Bowen Zhang, Erik Daxberger, Floris Weers, Marcin Eichner, Michael Emmersberger, Ruoming Pang, Tom Gunter, Xianzhi Du, Yinfei Yang","submitted_at":"2023-09-08T14:24:10Z","abstract_excerpt":"Sparse Mixture-of-Experts models (MoEs) have recently gained popularity due to their ability to decouple model size from inference efficiency by only activating a small subset of the model parameters for any given input token. As such, sparse MoEs have enabled unprecedented scalability, resulting in tremendous successes across domains such as natural language processing and computer vision. In this work, we instead explore the use of sparse MoEs to scale-down Vision Transformers (ViTs) to make them more attractive for resource-constrained vision applications. To this end, we propose a simplifi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.04354","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.04354/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.04354","created_at":"2026-07-05T06:48:58.691019+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.04354v1","created_at":"2026-07-05T06:48:58.691019+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.04354","created_at":"2026-07-05T06:48:58.691019+00:00"},{"alias_kind":"pith_short_12","alias_value":"WN2SAGEIELIW","created_at":"2026-07-05T06:48:58.691019+00:00"},{"alias_kind":"pith_short_16","alias_value":"WN2SAGEIELIWLF2O","created_at":"2026-07-05T06:48:58.691019+00:00"},{"alias_kind":"pith_short_8","alias_value":"WN2SAGEI","created_at":"2026-07-05T06:48:58.691019+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.20991","citing_title":"ExpertSim: Fast Particle Detector Simulation Using Mixture-of-Generative-Experts","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43","json":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43.json","graph_json":"https://pith.science/api/pith-number/WN2SAGEIELIWLF2ORF5HO2EY43/graph.json","events_json":"https://pith.science/api/pith-number/WN2SAGEIELIWLF2ORF5HO2EY43/events.json","paper":"https://pith.science/paper/WN2SAGEI"},"agent_actions":{"view_html":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43","download_json":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43.json","view_paper":"https://pith.science/paper/WN2SAGEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.04354&json=true","fetch_graph":"https://pith.science/api/pith-number/WN2SAGEIELIWLF2ORF5HO2EY43/graph.json","fetch_events":"https://pith.science/api/pith-number/WN2SAGEIELIWLF2ORF5HO2EY43/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43/action/storage_attestation","attest_author":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43/action/author_attestation","sign_citation":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43/action/citation_signature","submit_replication":"https://pith.science/pith/WN2SAGEIELIWLF2ORF5HO2EY43/action/replication_record"}},"created_at":"2026-07-05T06:48:58.691019+00:00","updated_at":"2026-07-05T06:48:58.691019+00:00"}