{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ELSNB42WFBVEC4DK2GFJJZMKTQ","short_pith_number":"pith:ELSNB42W","schema_version":"1.0","canonical_sha256":"22e4d0f356286a41706ad18a94e58a9c11491a5bf9d185bc8eeeca8766a941d0","source":{"kind":"arxiv","id":"2503.03613","version":1},"attestation_state":"computed","paper":{"title":"CLIP is Strong Enough to Fight Back: Test-time Counterattacks towards Zero-shot Adversarial Robustness of CLIP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Nicu Sebe, Songlong Xing, Zhengyu Zhao","submitted_at":"2025-03-05T15:51:59Z","abstract_excerpt":"Despite its prevalent use in image-text matching tasks in a zero-shot manner, CLIP has been shown to be highly vulnerable to adversarial perturbations added onto images. Recent studies propose to finetune the vision encoder of CLIP with adversarial samples generated on the fly, and show improved robustness against adversarial attacks on a spectrum of downstream datasets, a property termed as zero-shot robustness. In this paper, we show that malicious perturbations that seek to maximise the classification loss lead to `falsely stable' images, and propose to leverage the pre-trained vision encod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.03613","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-05T15:51:59Z","cross_cats_sorted":[],"title_canon_sha256":"9dcf4c6fc0a4bf05bfa1c6cbd0e47ccade8c454b1dbe37b1a713ef8385c84267","abstract_canon_sha256":"26d2fbe4ee28717fda1e481f375e97baec284fc6accd4906f707eb3dd4363bd3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:57.234830Z","signature_b64":"lY3j577MzqPbsmjIU9BYQs+rqSqx/+eFT9xudN0YM9bMhle1d4ev0MrSOOVrQpE3KlUh1m3Bn2CDf7oEhijkCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22e4d0f356286a41706ad18a94e58a9c11491a5bf9d185bc8eeeca8766a941d0","last_reissued_at":"2026-07-05T10:24:57.233956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:57.233956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CLIP is Strong Enough to Fight Back: Test-time Counterattacks towards Zero-shot Adversarial Robustness of CLIP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Nicu Sebe, Songlong Xing, Zhengyu Zhao","submitted_at":"2025-03-05T15:51:59Z","abstract_excerpt":"Despite its prevalent use in image-text matching tasks in a zero-shot manner, CLIP has been shown to be highly vulnerable to adversarial perturbations added onto images. Recent studies propose to finetune the vision encoder of CLIP with adversarial samples generated on the fly, and show improved robustness against adversarial attacks on a spectrum of downstream datasets, a property termed as zero-shot robustness. In this paper, we show that malicious perturbations that seek to maximise the classification loss lead to `falsely stable' images, and propose to leverage the pre-trained vision encod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.03613","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.03613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.03613","created_at":"2026-07-05T10:24:57.234067+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.03613v1","created_at":"2026-07-05T10:24:57.234067+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.03613","created_at":"2026-07-05T10:24:57.234067+00:00"},{"alias_kind":"pith_short_12","alias_value":"ELSNB42WFBVE","created_at":"2026-07-05T10:24:57.234067+00:00"},{"alias_kind":"pith_short_16","alias_value":"ELSNB42WFBVEC4DK","created_at":"2026-07-05T10:24:57.234067+00:00"},{"alias_kind":"pith_short_8","alias_value":"ELSNB42W","created_at":"2026-07-05T10:24:57.234067+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20366","citing_title":"Mitigating Hallucinations in Large Vision-Language Models without Performance Degradation","ref_index":130,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ","json":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ.json","graph_json":"https://pith.science/api/pith-number/ELSNB42WFBVEC4DK2GFJJZMKTQ/graph.json","events_json":"https://pith.science/api/pith-number/ELSNB42WFBVEC4DK2GFJJZMKTQ/events.json","paper":"https://pith.science/paper/ELSNB42W"},"agent_actions":{"view_html":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ","download_json":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ.json","view_paper":"https://pith.science/paper/ELSNB42W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.03613&json=true","fetch_graph":"https://pith.science/api/pith-number/ELSNB42WFBVEC4DK2GFJJZMKTQ/graph.json","fetch_events":"https://pith.science/api/pith-number/ELSNB42WFBVEC4DK2GFJJZMKTQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ/action/storage_attestation","attest_author":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ/action/author_attestation","sign_citation":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ/action/citation_signature","submit_replication":"https://pith.science/pith/ELSNB42WFBVEC4DK2GFJJZMKTQ/action/replication_record"}},"created_at":"2026-07-05T10:24:57.234067+00:00","updated_at":"2026-07-05T10:24:57.234067+00:00"}