{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UXR55S4LBQMOPT4PLYPTZ5LZ2U","short_pith_number":"pith:UXR55S4L","schema_version":"1.0","canonical_sha256":"a5e3decb8b0c18e7cf8f5e1f3cf579d53db2cf8c413a47efdfe76f9fc294b746","source":{"kind":"arxiv","id":"2502.11725","version":1},"attestation_state":"computed","paper":{"title":"Adversarially Robust CLIP Models Can Induce Better (Robust) Perceptual Metrics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Christian Schlarmann, Francesco Croce, Matthias Hein, Naman Deep Singh","submitted_at":"2025-02-17T12:11:01Z","abstract_excerpt":"Measuring perceptual similarity is a key tool in computer vision. In recent years perceptual metrics based on features extracted from neural networks with large and diverse training sets, e.g. CLIP, have become popular. At the same time, the metrics extracted from features of neural networks are not adversarially robust. In this paper we show that adversarially robust CLIP models, called R-CLIP$_\\textrm{F}$, obtained by unsupervised adversarial fine-tuning induce a better and adversarially robust perceptual metric that outperforms existing metrics in a zero-shot setting, and further matches th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11725","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-17T12:11:01Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"68f07e777eeb16168ed8ccd734d99ab0b21e583df1451eb17248044debd5460a","abstract_canon_sha256":"0eb2a77851de9368f6e2115cff6e159297c0209ee15d2dce6fa365eb0f4d4995"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:33.216946Z","signature_b64":"RmEi5S5mcFLZtmDMhXY6QVZ/PbOyD1KhYCcp4PSYcllOEecwNg8+HeaRXfdlQLHKyD0QksO97Aqpj2dmEfLgCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5e3decb8b0c18e7cf8f5e1f3cf579d53db2cf8c413a47efdfe76f9fc294b746","last_reissued_at":"2026-07-05T10:15:33.216383Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:33.216383Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adversarially Robust CLIP Models Can Induce Better (Robust) Perceptual Metrics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Christian Schlarmann, Francesco Croce, Matthias Hein, Naman Deep Singh","submitted_at":"2025-02-17T12:11:01Z","abstract_excerpt":"Measuring perceptual similarity is a key tool in computer vision. In recent years perceptual metrics based on features extracted from neural networks with large and diverse training sets, e.g. CLIP, have become popular. At the same time, the metrics extracted from features of neural networks are not adversarially robust. In this paper we show that adversarially robust CLIP models, called R-CLIP$_\\textrm{F}$, obtained by unsupervised adversarial fine-tuning induce a better and adversarially robust perceptual metric that outperforms existing metrics in a zero-shot setting, and further matches th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11725","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11725/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11725","created_at":"2026-07-05T10:15:33.216441+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11725v1","created_at":"2026-07-05T10:15:33.216441+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11725","created_at":"2026-07-05T10:15:33.216441+00:00"},{"alias_kind":"pith_short_12","alias_value":"UXR55S4LBQMO","created_at":"2026-07-05T10:15:33.216441+00:00"},{"alias_kind":"pith_short_16","alias_value":"UXR55S4LBQMOPT4P","created_at":"2026-07-05T10:15:33.216441+00:00"},{"alias_kind":"pith_short_8","alias_value":"UXR55S4L","created_at":"2026-07-05T10:15:33.216441+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.14204","citing_title":"Beginning with You: Perceptual-Initialization Improves Vision-Language Representation and Alignment","ref_index":26,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U","json":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U.json","graph_json":"https://pith.science/api/pith-number/UXR55S4LBQMOPT4PLYPTZ5LZ2U/graph.json","events_json":"https://pith.science/api/pith-number/UXR55S4LBQMOPT4PLYPTZ5LZ2U/events.json","paper":"https://pith.science/paper/UXR55S4L"},"agent_actions":{"view_html":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U","download_json":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U.json","view_paper":"https://pith.science/paper/UXR55S4L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11725&json=true","fetch_graph":"https://pith.science/api/pith-number/UXR55S4LBQMOPT4PLYPTZ5LZ2U/graph.json","fetch_events":"https://pith.science/api/pith-number/UXR55S4LBQMOPT4PLYPTZ5LZ2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U/action/storage_attestation","attest_author":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U/action/author_attestation","sign_citation":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U/action/citation_signature","submit_replication":"https://pith.science/pith/UXR55S4LBQMOPT4PLYPTZ5LZ2U/action/replication_record"}},"created_at":"2026-07-05T10:15:33.216441+00:00","updated_at":"2026-07-05T10:15:33.216441+00:00"}