{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ULC63RNSO76MMSUX2QQ57QEC2M","short_pith_number":"pith:ULC63RNS","schema_version":"1.0","canonical_sha256":"a2c5edc5b277fcc64a97d421dfc082d30769fc53b860081d16ee7e2595efb159","source":{"kind":"arxiv","id":"2303.14465","version":2},"attestation_state":"computed","paper":{"title":"Equivariant Similarity for Vision-Language Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chung-Ching Lin, Hanwang Zhang, Kevin Lin, Lijuan Wang, Linjie Li, Tan Wang, Zhengyuan Yang, Zicheng Liu","submitted_at":"2023-03-25T13:22:56Z","abstract_excerpt":"This study explores the concept of equivariance in vision-language foundation models (VLMs), focusing specifically on the multimodal similarity function that is not only the major training objective but also the core delivery to support downstream tasks. Unlike the existing image-text similarity objective which only categorizes matched pairs as similar and unmatched pairs as dissimilar, equivariance also requires similarity to vary faithfully according to the semantic changes. This allows VLMs to generalize better to nuanced and unseen multimodal compositions. However, modeling equivariance is"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.14465","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-25T13:22:56Z","cross_cats_sorted":[],"title_canon_sha256":"8835d712e89d5778c5cde31b421a0b3a0c505da1ebb339105a746b8659308a85","abstract_canon_sha256":"e3a130ec2783a5beb483f07e11f036c599a4a13b2ac6c0dbb9a5dbbe29022b7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:39.205855Z","signature_b64":"lrf0WM2GzbkzWmPRqlTfMGFlpkIWDy8ErFyDu7Poizp2nzdGJY/LvopetDbnl59eXxT3UZ5X8GJIIcVQwK9eDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2c5edc5b277fcc64a97d421dfc082d30769fc53b860081d16ee7e2595efb159","last_reissued_at":"2026-07-05T06:58:39.205358Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:39.205358Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Equivariant Similarity for Vision-Language Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chung-Ching Lin, Hanwang Zhang, Kevin Lin, Lijuan Wang, Linjie Li, Tan Wang, Zhengyuan Yang, Zicheng Liu","submitted_at":"2023-03-25T13:22:56Z","abstract_excerpt":"This study explores the concept of equivariance in vision-language foundation models (VLMs), focusing specifically on the multimodal similarity function that is not only the major training objective but also the core delivery to support downstream tasks. Unlike the existing image-text similarity objective which only categorizes matched pairs as similar and unmatched pairs as dissimilar, equivariance also requires similarity to vary faithfully according to the semantic changes. This allows VLMs to generalize better to nuanced and unseen multimodal compositions. However, modeling equivariance is"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.14465","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.14465/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.14465","created_at":"2026-07-05T06:58:39.205424+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.14465v2","created_at":"2026-07-05T06:58:39.205424+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.14465","created_at":"2026-07-05T06:58:39.205424+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULC63RNSO76M","created_at":"2026-07-05T06:58:39.205424+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULC63RNSO76MMSUX","created_at":"2026-07-05T06:58:39.205424+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULC63RNS","created_at":"2026-07-05T06:58:39.205424+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.09906","citing_title":"Insect-Foundation: A Foundation Model and Large Multimodal Dataset for Vision-Language Insect Understanding","ref_index":65,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M","json":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M.json","graph_json":"https://pith.science/api/pith-number/ULC63RNSO76MMSUX2QQ57QEC2M/graph.json","events_json":"https://pith.science/api/pith-number/ULC63RNSO76MMSUX2QQ57QEC2M/events.json","paper":"https://pith.science/paper/ULC63RNS"},"agent_actions":{"view_html":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M","download_json":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M.json","view_paper":"https://pith.science/paper/ULC63RNS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.14465&json=true","fetch_graph":"https://pith.science/api/pith-number/ULC63RNSO76MMSUX2QQ57QEC2M/graph.json","fetch_events":"https://pith.science/api/pith-number/ULC63RNSO76MMSUX2QQ57QEC2M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M/action/storage_attestation","attest_author":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M/action/author_attestation","sign_citation":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M/action/citation_signature","submit_replication":"https://pith.science/pith/ULC63RNSO76MMSUX2QQ57QEC2M/action/replication_record"}},"created_at":"2026-07-05T06:58:39.205424+00:00","updated_at":"2026-07-05T06:58:39.205424+00:00"}