{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UEFROMZ5RD65RHZ4IIBXVJU3UD","short_pith_number":"pith:UEFROMZ5","schema_version":"1.0","canonical_sha256":"a10b17333d88fdd89f3c42037aa69ba0d0a10019e934b07ef7698e024a13e1f4","source":{"kind":"arxiv","id":"2303.07226","version":1},"attestation_state":"computed","paper":{"title":"Scaling Vision-Language Models with Sparse Mixture of Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chunyuan Li, Kurt Keutzer, Sheng Shen, Trevor Darrell, Yuxiong He, Zhewei Yao","submitted_at":"2023-03-13T16:00:31Z","abstract_excerpt":"The field of natural language processing (NLP) has made significant strides in recent years, particularly in the development of large-scale vision-language models (VLMs). These models aim to bridge the gap between text and visual information, enabling a more comprehensive understanding of multimedia data. However, as these models become larger and more complex, they also become more challenging to train and deploy. One approach to addressing this challenge is the use of sparsely-gated mixture-of-experts (MoE) techniques, which divide the model into smaller, specialized sub-models that can join"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.07226","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-03-13T16:00:31Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"99e993cde0f757825a001f6ff70ca374e936482b468d92452350254f9726e19c","abstract_canon_sha256":"d41dc77d984dcd9fda45bc8cd9d92bbc1850245fee1c63f91276cf7cfbb35e41"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:50:33.922444Z","signature_b64":"39KQblNqhEcGVh4OEN8nBwHHGQb8f8xvuB9LVxG5JL8pCVFezOwg9aFkOCnjlz1ee65cObC0QfIISS63f2YpBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a10b17333d88fdd89f3c42037aa69ba0d0a10019e934b07ef7698e024a13e1f4","last_reissued_at":"2026-07-05T05:50:33.921892Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:50:33.921892Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Vision-Language Models with Sparse Mixture of Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chunyuan Li, Kurt Keutzer, Sheng Shen, Trevor Darrell, Yuxiong He, Zhewei Yao","submitted_at":"2023-03-13T16:00:31Z","abstract_excerpt":"The field of natural language processing (NLP) has made significant strides in recent years, particularly in the development of large-scale vision-language models (VLMs). These models aim to bridge the gap between text and visual information, enabling a more comprehensive understanding of multimedia data. However, as these models become larger and more complex, they also become more challenging to train and deploy. One approach to addressing this challenge is the use of sparsely-gated mixture-of-experts (MoE) techniques, which divide the model into smaller, specialized sub-models that can join"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.07226","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.07226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.07226","created_at":"2026-07-05T05:50:33.921950+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.07226v1","created_at":"2026-07-05T05:50:33.921950+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.07226","created_at":"2026-07-05T05:50:33.921950+00:00"},{"alias_kind":"pith_short_12","alias_value":"UEFROMZ5RD65","created_at":"2026-07-05T05:50:33.921950+00:00"},{"alias_kind":"pith_short_16","alias_value":"UEFROMZ5RD65RHZ4","created_at":"2026-07-05T05:50:33.921950+00:00"},{"alias_kind":"pith_short_8","alias_value":"UEFROMZ5","created_at":"2026-07-05T05:50:33.921950+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21645","citing_title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00275","citing_title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20610","citing_title":"Beyond Routing: Characterising Expert Tuning and Representation in Vision Mixture-of-Experts","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2401.15947","citing_title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD","json":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD.json","graph_json":"https://pith.science/api/pith-number/UEFROMZ5RD65RHZ4IIBXVJU3UD/graph.json","events_json":"https://pith.science/api/pith-number/UEFROMZ5RD65RHZ4IIBXVJU3UD/events.json","paper":"https://pith.science/paper/UEFROMZ5"},"agent_actions":{"view_html":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD","download_json":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD.json","view_paper":"https://pith.science/paper/UEFROMZ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.07226&json=true","fetch_graph":"https://pith.science/api/pith-number/UEFROMZ5RD65RHZ4IIBXVJU3UD/graph.json","fetch_events":"https://pith.science/api/pith-number/UEFROMZ5RD65RHZ4IIBXVJU3UD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD/action/storage_attestation","attest_author":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD/action/author_attestation","sign_citation":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD/action/citation_signature","submit_replication":"https://pith.science/pith/UEFROMZ5RD65RHZ4IIBXVJU3UD/action/replication_record"}},"created_at":"2026-07-05T05:50:33.921950+00:00","updated_at":"2026-07-05T05:50:33.921950+00:00"}