{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MV7BGFY4ZMAOR7YYOSWWISRJDD","short_pith_number":"pith:MV7BGFY4","schema_version":"1.0","canonical_sha256":"657e13171ccb00e8ff1874ad644a2918f0de5230b5d6a87e1f866d902512d347","source":{"kind":"arxiv","id":"2503.00329","version":2},"attestation_state":"computed","paper":{"title":"ABC: Achieving Better Control of Multimodal Embeddings using VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Benjamin Schneider, Florian Kerschbaum, Wenhu Chen","submitted_at":"2025-03-01T03:29:02Z","abstract_excerpt":"Visual embedding models excel at zero-shot tasks like visual retrieval and classification. However, these models cannot be used for tasks that contain ambiguity or require user instruction. These tasks necessitate an embedding model which outputs can use a natural language instruction to control the representation of a visual embedding. Existing CLIP-based approaches embed images and text independently, and fuse the result. We find that this results in weak interactions between modalities, and poor user control over the representation. We introduce ABC, an open-source multimodal embedding mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.00329","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-01T03:29:02Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c4061681e60677925defa6d65b6f238a5e3ca7b43eeee9abb195998de7a6801e","abstract_canon_sha256":"ef8d2f6591ae460b16b850f1940de86e72dd9eb55b4ee77ad33c5db5466e5c36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:53.398811Z","signature_b64":"mmi/bGcLFPkOGhmrDrtGiMC6Mr+wNFtRvYPOQm4fCnfjYc18QjA3PMBkhh56Htu9ST327DUPcYF06IbqqaniDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"657e13171ccb00e8ff1874ad644a2918f0de5230b5d6a87e1f866d902512d347","last_reissued_at":"2026-07-05T11:56:53.398338Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:53.398338Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ABC: Achieving Better Control of Multimodal Embeddings using VLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Benjamin Schneider, Florian Kerschbaum, Wenhu Chen","submitted_at":"2025-03-01T03:29:02Z","abstract_excerpt":"Visual embedding models excel at zero-shot tasks like visual retrieval and classification. However, these models cannot be used for tasks that contain ambiguity or require user instruction. These tasks necessitate an embedding model which outputs can use a natural language instruction to control the representation of a visual embedding. Existing CLIP-based approaches embed images and text independently, and fuse the result. We find that this results in weak interactions between modalities, and poor user control over the representation. We introduce ABC, an open-source multimodal embedding mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.00329","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.00329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.00329","created_at":"2026-07-05T11:56:53.398397+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.00329v2","created_at":"2026-07-05T11:56:53.398397+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.00329","created_at":"2026-07-05T11:56:53.398397+00:00"},{"alias_kind":"pith_short_12","alias_value":"MV7BGFY4ZMAO","created_at":"2026-07-05T11:56:53.398397+00:00"},{"alias_kind":"pith_short_16","alias_value":"MV7BGFY4ZMAOR7YY","created_at":"2026-07-05T11:56:53.398397+00:00"},{"alias_kind":"pith_short_8","alias_value":"MV7BGFY4","created_at":"2026-07-05T11:56:53.398397+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24621","citing_title":"FreeRet: MLLMs as Training-Free Retrievers","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD","json":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD.json","graph_json":"https://pith.science/api/pith-number/MV7BGFY4ZMAOR7YYOSWWISRJDD/graph.json","events_json":"https://pith.science/api/pith-number/MV7BGFY4ZMAOR7YYOSWWISRJDD/events.json","paper":"https://pith.science/paper/MV7BGFY4"},"agent_actions":{"view_html":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD","download_json":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD.json","view_paper":"https://pith.science/paper/MV7BGFY4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.00329&json=true","fetch_graph":"https://pith.science/api/pith-number/MV7BGFY4ZMAOR7YYOSWWISRJDD/graph.json","fetch_events":"https://pith.science/api/pith-number/MV7BGFY4ZMAOR7YYOSWWISRJDD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD/action/storage_attestation","attest_author":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD/action/author_attestation","sign_citation":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD/action/citation_signature","submit_replication":"https://pith.science/pith/MV7BGFY4ZMAOR7YYOSWWISRJDD/action/replication_record"}},"created_at":"2026-07-05T11:56:53.398397+00:00","updated_at":"2026-07-05T11:56:53.398397+00:00"}