{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4T5MQ6NX3TDZPMIJ45K3IZTS37","short_pith_number":"pith:4T5MQ6NX","schema_version":"1.0","canonical_sha256":"e4fac879b7dcc797b109e755b46672dffab48a7002be57b939ef9034a3f6b466","source":{"kind":"arxiv","id":"2407.17083","version":1},"attestation_state":"computed","paper":{"title":"When Text and Images Don't Mix: Bias-Correcting Language-Image Similarity Scores for Anomaly Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Adam Goodge, Bryan Hooi, Wee Siong Ng","submitted_at":"2024-07-24T08:20:02Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) achieves remarkable performance in various downstream tasks through the alignment of image and text input embeddings and holds great promise for anomaly detection. However, our empirical experiments show that the embeddings of text inputs unexpectedly tightly cluster together, far away from image embeddings, contrary to the model's contrastive training objective to align image-text input pairs. We show that this phenomenon induces a `similarity bias' - in which false negative and false positive errors occur due to bias in the similarities between "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.17083","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-24T08:20:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f28296bd85aaeb2bbfe34ef50c46ad6bf3bdbaff5bbb03f2ac88e705434e1acc","abstract_canon_sha256":"8445fb5ae68acdb89780c34a954601e714d696a534a7dd3aca7d06b13ee73ead"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:48:01.895583Z","signature_b64":"d3VNmDTVauuorZVwUEx0P+SI/I/2F2aWttNSipTdO5blrKDkwS9nWgjbR0OyOTS0lS500vMJ0hLV30M4wYzqBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e4fac879b7dcc797b109e755b46672dffab48a7002be57b939ef9034a3f6b466","last_reissued_at":"2026-07-05T08:48:01.895220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:48:01.895220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Text and Images Don't Mix: Bias-Correcting Language-Image Similarity Scores for Anomaly Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Adam Goodge, Bryan Hooi, Wee Siong Ng","submitted_at":"2024-07-24T08:20:02Z","abstract_excerpt":"Contrastive Language-Image Pre-training (CLIP) achieves remarkable performance in various downstream tasks through the alignment of image and text input embeddings and holds great promise for anomaly detection. However, our empirical experiments show that the embeddings of text inputs unexpectedly tightly cluster together, far away from image embeddings, contrary to the model's contrastive training objective to align image-text input pairs. We show that this phenomenon induces a `similarity bias' - in which false negative and false positive errors occur due to bias in the similarities between "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.17083","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.17083/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.17083","created_at":"2026-07-05T08:48:01.895282+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.17083v1","created_at":"2026-07-05T08:48:01.895282+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.17083","created_at":"2026-07-05T08:48:01.895282+00:00"},{"alias_kind":"pith_short_12","alias_value":"4T5MQ6NX3TDZ","created_at":"2026-07-05T08:48:01.895282+00:00"},{"alias_kind":"pith_short_16","alias_value":"4T5MQ6NX3TDZPMIJ","created_at":"2026-07-05T08:48:01.895282+00:00"},{"alias_kind":"pith_short_8","alias_value":"4T5MQ6NX","created_at":"2026-07-05T08:48:01.895282+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.21892","citing_title":"SODA: Out-of-Distribution Detection in Domain-Shifted Point Clouds via Neighborhood Propagation","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37","json":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37.json","graph_json":"https://pith.science/api/pith-number/4T5MQ6NX3TDZPMIJ45K3IZTS37/graph.json","events_json":"https://pith.science/api/pith-number/4T5MQ6NX3TDZPMIJ45K3IZTS37/events.json","paper":"https://pith.science/paper/4T5MQ6NX"},"agent_actions":{"view_html":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37","download_json":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37.json","view_paper":"https://pith.science/paper/4T5MQ6NX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.17083&json=true","fetch_graph":"https://pith.science/api/pith-number/4T5MQ6NX3TDZPMIJ45K3IZTS37/graph.json","fetch_events":"https://pith.science/api/pith-number/4T5MQ6NX3TDZPMIJ45K3IZTS37/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37/action/storage_attestation","attest_author":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37/action/author_attestation","sign_citation":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37/action/citation_signature","submit_replication":"https://pith.science/pith/4T5MQ6NX3TDZPMIJ45K3IZTS37/action/replication_record"}},"created_at":"2026-07-05T08:48:01.895282+00:00","updated_at":"2026-07-05T08:48:01.895282+00:00"}