{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QWFWV6IQSK2WRBU5T6DFYQ4SJ5","short_pith_number":"pith:QWFWV6IQ","schema_version":"1.0","canonical_sha256":"858b6af91092b568869d9f865c43924f772876406052c8191b897b464e328c4b","source":{"kind":"arxiv","id":"2409.09721","version":2},"attestation_state":"computed","paper":{"title":"Finetuning CLIP to Reason about Pairwise Differences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Devin Willmott, Dylan Sam, Joao D. Semedo, J. Zico Kolter","submitted_at":"2024-09-15T13:02:14Z","abstract_excerpt":"Vision-language models (VLMs) such as CLIP are trained via contrastive learning between text and image pairs, resulting in aligned image and text embeddings that are useful for many downstream tasks. A notable drawback of CLIP, however, is that the resulting embedding space seems to lack some of the structure of its purely text-based alternatives. For instance, while text embeddings have long been noted to satisfy analogies in embedding space using vector arithmetic, CLIP has no such property. In this paper, we propose an approach to natively train CLIP in a contrastive manner to reason about "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.09721","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-15T13:02:14Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"fca7048b12c27277f66292391f4a06416f3fa863adf372b906b1acd7b5e2b34f","abstract_canon_sha256":"55a22224adfa984edf3512825f46633a6d4f53899b5fbd37eb7f38f4a7611501"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:39.992870Z","signature_b64":"F+JohOlYk4cuRF2xXhBxhSKyUkbkhqMQi8OVcTWsslSaIMOydUYuGYR6CGA3/Ra5jLMbXHKCZczXasJqBLmNDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"858b6af91092b568869d9f865c43924f772876406052c8191b897b464e328c4b","last_reissued_at":"2026-07-05T11:31:39.992344Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:39.992344Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Finetuning CLIP to Reason about Pairwise Differences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Devin Willmott, Dylan Sam, Joao D. Semedo, J. Zico Kolter","submitted_at":"2024-09-15T13:02:14Z","abstract_excerpt":"Vision-language models (VLMs) such as CLIP are trained via contrastive learning between text and image pairs, resulting in aligned image and text embeddings that are useful for many downstream tasks. A notable drawback of CLIP, however, is that the resulting embedding space seems to lack some of the structure of its purely text-based alternatives. For instance, while text embeddings have long been noted to satisfy analogies in embedding space using vector arithmetic, CLIP has no such property. In this paper, we propose an approach to natively train CLIP in a contrastive manner to reason about "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.09721","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.09721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.09721","created_at":"2026-07-05T11:31:39.992403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.09721v2","created_at":"2026-07-05T11:31:39.992403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.09721","created_at":"2026-07-05T11:31:39.992403+00:00"},{"alias_kind":"pith_short_12","alias_value":"QWFWV6IQSK2W","created_at":"2026-07-05T11:31:39.992403+00:00"},{"alias_kind":"pith_short_16","alias_value":"QWFWV6IQSK2WRBU5","created_at":"2026-07-05T11:31:39.992403+00:00"},{"alias_kind":"pith_short_8","alias_value":"QWFWV6IQ","created_at":"2026-07-05T11:31:39.992403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.23370","citing_title":"GRAPE: Let GRPO Supervise Query Rewriting by Ranking for Retrieval","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5","json":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5.json","graph_json":"https://pith.science/api/pith-number/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/graph.json","events_json":"https://pith.science/api/pith-number/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/events.json","paper":"https://pith.science/paper/QWFWV6IQ"},"agent_actions":{"view_html":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5","download_json":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5.json","view_paper":"https://pith.science/paper/QWFWV6IQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.09721&json=true","fetch_graph":"https://pith.science/api/pith-number/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/graph.json","fetch_events":"https://pith.science/api/pith-number/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/action/storage_attestation","attest_author":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/action/author_attestation","sign_citation":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/action/citation_signature","submit_replication":"https://pith.science/pith/QWFWV6IQSK2WRBU5T6DFYQ4SJ5/action/replication_record"}},"created_at":"2026-07-05T11:31:39.992403+00:00","updated_at":"2026-07-05T11:31:39.992403+00:00"}