{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BR73RUM7XRJK56HX73QGAUIYRS","short_pith_number":"pith:BR73RUM7","schema_version":"1.0","canonical_sha256":"0c7fb8d19fbc52aef8f7fee06051188c89c11603a70b0518388828a54c20bcd9","source":{"kind":"arxiv","id":"2406.02915","version":1},"attestation_state":"computed","paper":{"title":"Visual-Text Cross Alignment: Refining the Similarity Score in Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Feng Liu, Haopeng Li, James Bailey, Jinhao Li, Lei Feng, Sarah Erfani","submitted_at":"2024-06-05T04:08:41Z","abstract_excerpt":"It has recently been discovered that using a pre-trained vision-language model (VLM), e.g., CLIP, to align a whole query image with several finer text descriptions generated by a large language model can significantly enhance zero-shot performance. However, in this paper, we empirically find that the finer descriptions tend to align more effectively with local areas of the query image rather than the whole image, and then we theoretically validate this finding. Thus, we present a method called weighted visual-text cross alignment (WCA). This method begins with a localized visual prompting tech"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.02915","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-05T04:08:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a7ec9b3c905d6bc282077c1c92384314af1542a17519e6f6ce65c7aca8faf4d3","abstract_canon_sha256":"0276aebaf344618cc96b5455a3ab24c82eb29c9804322835bc6b96ac2a82ace8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:32.624786Z","signature_b64":"5nQZDMjGRRqF63KEdfEGesGsAY+yQz9EJfOBorQh3S+49vsAgNkzzQnmTUPO/ITeia+iLQuFO50CWgP5WJN2BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c7fb8d19fbc52aef8f7fee06051188c89c11603a70b0518388828a54c20bcd9","last_reissued_at":"2026-07-05T08:27:32.624246Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:32.624246Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual-Text Cross Alignment: Refining the Similarity Score in Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Feng Liu, Haopeng Li, James Bailey, Jinhao Li, Lei Feng, Sarah Erfani","submitted_at":"2024-06-05T04:08:41Z","abstract_excerpt":"It has recently been discovered that using a pre-trained vision-language model (VLM), e.g., CLIP, to align a whole query image with several finer text descriptions generated by a large language model can significantly enhance zero-shot performance. However, in this paper, we empirically find that the finer descriptions tend to align more effectively with local areas of the query image rather than the whole image, and then we theoretically validate this finding. Thus, we present a method called weighted visual-text cross alignment (WCA). This method begins with a localized visual prompting tech"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02915","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.02915","created_at":"2026-07-05T08:27:32.624306+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.02915v1","created_at":"2026-07-05T08:27:32.624306+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02915","created_at":"2026-07-05T08:27:32.624306+00:00"},{"alias_kind":"pith_short_12","alias_value":"BR73RUM7XRJK","created_at":"2026-07-05T08:27:32.624306+00:00"},{"alias_kind":"pith_short_16","alias_value":"BR73RUM7XRJK56HX","created_at":"2026-07-05T08:27:32.624306+00:00"},{"alias_kind":"pith_short_8","alias_value":"BR73RUM7","created_at":"2026-07-05T08:27:32.624306+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.17417","citing_title":"Constrained Prompt Enhancement for Improving Zero-Shot Generalization of Vision-Language Models","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS","json":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS.json","graph_json":"https://pith.science/api/pith-number/BR73RUM7XRJK56HX73QGAUIYRS/graph.json","events_json":"https://pith.science/api/pith-number/BR73RUM7XRJK56HX73QGAUIYRS/events.json","paper":"https://pith.science/paper/BR73RUM7"},"agent_actions":{"view_html":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS","download_json":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS.json","view_paper":"https://pith.science/paper/BR73RUM7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.02915&json=true","fetch_graph":"https://pith.science/api/pith-number/BR73RUM7XRJK56HX73QGAUIYRS/graph.json","fetch_events":"https://pith.science/api/pith-number/BR73RUM7XRJK56HX73QGAUIYRS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS/action/storage_attestation","attest_author":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS/action/author_attestation","sign_citation":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS/action/citation_signature","submit_replication":"https://pith.science/pith/BR73RUM7XRJK56HX73QGAUIYRS/action/replication_record"}},"created_at":"2026-07-05T08:27:32.624306+00:00","updated_at":"2026-07-05T08:27:32.624306+00:00"}