{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EWNQKNTNJ72L4UKFU34ZIAV5O7","short_pith_number":"pith:EWNQKNTN","schema_version":"1.0","canonical_sha256":"259b05366d4ff4be5145a6f99402bd77c06889e61219184aaec774ebde9b640f","source":{"kind":"arxiv","id":"2409.19291","version":3},"attestation_state":"computed","paper":{"title":"CLIP-MoE: Towards Building Mixture of Experts for CLIP with Diversified Multiplet Upcycling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jihai Zhang, Tong Zhu, Xiaoye Qu, Yu Cheng","submitted_at":"2024-09-28T09:28:51Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) has become a cornerstone in multimodal intelligence. However, recent studies discovered that CLIP can only encode one aspect of the feature space, leading to substantial information loss and indistinctive features. To mitigate this issue, this paper introduces a novel strategy that fine-tunes a series of complementary CLIP models and transforms them into a CLIP-MoE. Specifically, we propose a model-agnostic Diversified Multiplet Upcycling (DMU) framework for CLIP. Instead of training multiple CLIP models from scratch, DMU leverages a pre-trained C"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19291","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-28T09:28:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1a61cb33bb27c9a8e43b5a9754628fca6d9b125433f5c87f3dc5301755010f49","abstract_canon_sha256":"1c47b74c8026017420415879a8cf258c5f72b9ad61bd4ef9b7184768bb42d65c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:47.595611Z","signature_b64":"zqvVBAKsM93lBM3aAOOC1Lv5nbenimTEmzvOIiQkaQKWiOEm9lXPmVJSwPWZbsR1wwX4Wzc3R7xLE3KNUVQgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"259b05366d4ff4be5145a6f99402bd77c06889e61219184aaec774ebde9b640f","last_reissued_at":"2026-07-05T11:10:47.595062Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:47.595062Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIP-MoE: Towards Building Mixture of Experts for CLIP with Diversified Multiplet Upcycling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jihai Zhang, Tong Zhu, Xiaoye Qu, Yu Cheng","submitted_at":"2024-09-28T09:28:51Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) has become a cornerstone in multimodal intelligence. However, recent studies discovered that CLIP can only encode one aspect of the feature space, leading to substantial information loss and indistinctive features. To mitigate this issue, this paper introduces a novel strategy that fine-tunes a series of complementary CLIP models and transforms them into a CLIP-MoE. Specifically, we propose a model-agnostic Diversified Multiplet Upcycling (DMU) framework for CLIP. Instead of training multiple CLIP models from scratch, DMU leverages a pre-trained C"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19291","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19291","created_at":"2026-07-05T11:10:47.595128+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19291v3","created_at":"2026-07-05T11:10:47.595128+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19291","created_at":"2026-07-05T11:10:47.595128+00:00"},{"alias_kind":"pith_short_12","alias_value":"EWNQKNTNJ72L","created_at":"2026-07-05T11:10:47.595128+00:00"},{"alias_kind":"pith_short_16","alias_value":"EWNQKNTNJ72L4UKF","created_at":"2026-07-05T11:10:47.595128+00:00"},{"alias_kind":"pith_short_8","alias_value":"EWNQKNTN","created_at":"2026-07-05T11:10:47.595128+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.32040","citing_title":"FaceMoE: Mixture of Experts for Low-Resolution Face Recognition","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7","json":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7.json","graph_json":"https://pith.science/api/pith-number/EWNQKNTNJ72L4UKFU34ZIAV5O7/graph.json","events_json":"https://pith.science/api/pith-number/EWNQKNTNJ72L4UKFU34ZIAV5O7/events.json","paper":"https://pith.science/paper/EWNQKNTN"},"agent_actions":{"view_html":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7","download_json":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7.json","view_paper":"https://pith.science/paper/EWNQKNTN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19291&json=true","fetch_graph":"https://pith.science/api/pith-number/EWNQKNTNJ72L4UKFU34ZIAV5O7/graph.json","fetch_events":"https://pith.science/api/pith-number/EWNQKNTNJ72L4UKFU34ZIAV5O7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7/action/storage_attestation","attest_author":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7/action/author_attestation","sign_citation":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7/action/citation_signature","submit_replication":"https://pith.science/pith/EWNQKNTNJ72L4UKFU34ZIAV5O7/action/replication_record"}},"created_at":"2026-07-05T11:10:47.595128+00:00","updated_at":"2026-07-05T11:10:47.595128+00:00"}