{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OGFSX3DWC34YJBWR4SGT2EXYFK","short_pith_number":"pith:OGFSX3DW","schema_version":"1.0","canonical_sha256":"718b2bec7616f98486d1e48d3d12f82a85897243029a63382bc8663ce8c2a802","source":{"kind":"arxiv","id":"2402.12336","version":2},"attestation_state":"computed","paper":{"title":"Robust CLIP: Unsupervised Adversarial Fine-Tuning of Vision Embeddings for Robust Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Christian Schlarmann, Francesco Croce, Matthias Hein, Naman Deep Singh","submitted_at":"2024-02-19T18:09:48Z","abstract_excerpt":"Multi-modal foundation models like OpenFlamingo, LLaVA, and GPT-4 are increasingly used for various real-world tasks. Prior work has shown that these models are highly vulnerable to adversarial attacks on the vision modality. These attacks can be leveraged to spread fake information or defraud users, and thus pose a significant risk, which makes the robustness of large multi-modal foundation models a pressing problem. The CLIP model, or one of its variants, is used as a frozen vision encoder in many large vision-language models (LVLMs), e.g. LLaVA and OpenFlamingo. We propose an unsupervised a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12336","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-19T18:09:48Z","cross_cats_sorted":["cs.AI","cs.CV","stat.ML"],"title_canon_sha256":"6568c0df4a83ea7ee1d81ec5d20c22b94b936b3bd1c4575ddc38be4f7ef62ab1","abstract_canon_sha256":"001aaa0e65f2a0de61bfb0b8faf36b22cdd7b271f83429ab2dbfcf41b4ed0462"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:46.372731Z","signature_b64":"A5I/JScUJvKKPfKD9dPS3n00iRHTxwTe962rKYvU8POoB2/YDm21Sy0t00gbKpezwT7/gK1DmP9H53hrhy3XBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"718b2bec7616f98486d1e48d3d12f82a85897243029a63382bc8663ce8c2a802","last_reissued_at":"2026-07-05T08:27:46.372224Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:46.372224Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust CLIP: Unsupervised Adversarial Fine-Tuning of Vision Embeddings for Robust Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","stat.ML"],"primary_cat":"cs.LG","authors_text":"Christian Schlarmann, Francesco Croce, Matthias Hein, Naman Deep Singh","submitted_at":"2024-02-19T18:09:48Z","abstract_excerpt":"Multi-modal foundation models like OpenFlamingo, LLaVA, and GPT-4 are increasingly used for various real-world tasks. Prior work has shown that these models are highly vulnerable to adversarial attacks on the vision modality. These attacks can be leveraged to spread fake information or defraud users, and thus pose a significant risk, which makes the robustness of large multi-modal foundation models a pressing problem. The CLIP model, or one of its variants, is used as a frozen vision encoder in many large vision-language models (LVLMs), e.g. LLaVA and OpenFlamingo. We propose an unsupervised a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12336","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12336/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12336","created_at":"2026-07-05T08:27:46.372290+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12336v2","created_at":"2026-07-05T08:27:46.372290+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12336","created_at":"2026-07-05T08:27:46.372290+00:00"},{"alias_kind":"pith_short_12","alias_value":"OGFSX3DWC34Y","created_at":"2026-07-05T08:27:46.372290+00:00"},{"alias_kind":"pith_short_16","alias_value":"OGFSX3DWC34YJBWR","created_at":"2026-07-05T08:27:46.372290+00:00"},{"alias_kind":"pith_short_8","alias_value":"OGFSX3DW","created_at":"2026-07-05T08:27:46.372290+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18839","citing_title":"Semantic Robustness Certification for Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03793","citing_title":"Exploring Adversarial Robustness and Safety Alignment in Multilingual Multi-Modal Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03730","citing_title":"Beyond False Stability: High-Noise Drift Gating for Test-Time Adversarial Defenses in Vision-Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03713","citing_title":"Investigating Adversarial Robustness of Multi-modal Large Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25922","citing_title":"Closed-Loop Bidirectional Prompting for Adversarial Robustness of Vision Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2509.15435","citing_title":"ORCA: An Agentic Reasoning Framework for Hallucination and Adversarial Robustness in Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15584","citing_title":"AGC: Adaptive Geodesic Correction for Adversarial Robustness on Vision-Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2509.15435","citing_title":"ORCA: An Agentic Reasoning Framework for Hallucination and Adversarial Robustness in Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21893","citing_title":"Breaking the Illusion: Consensus-Based Generative Mitigation of Adversarial Illusions in Multi-Modal Embeddings","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07222","citing_title":"Pay Less Attention to Function Words for Free Robustness of Vision-Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08440","citing_title":"TARO: Temporal Adversarial Rectification Optimization Using Diffusion Models as Purifiers","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01449","citing_title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18867","citing_title":"Hierarchically Robust Zero-shot Vision-language Models","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK","json":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK.json","graph_json":"https://pith.science/api/pith-number/OGFSX3DWC34YJBWR4SGT2EXYFK/graph.json","events_json":"https://pith.science/api/pith-number/OGFSX3DWC34YJBWR4SGT2EXYFK/events.json","paper":"https://pith.science/paper/OGFSX3DW"},"agent_actions":{"view_html":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK","download_json":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK.json","view_paper":"https://pith.science/paper/OGFSX3DW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12336&json=true","fetch_graph":"https://pith.science/api/pith-number/OGFSX3DWC34YJBWR4SGT2EXYFK/graph.json","fetch_events":"https://pith.science/api/pith-number/OGFSX3DWC34YJBWR4SGT2EXYFK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK/action/storage_attestation","attest_author":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK/action/author_attestation","sign_citation":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK/action/citation_signature","submit_replication":"https://pith.science/pith/OGFSX3DWC34YJBWR4SGT2EXYFK/action/replication_record"}},"created_at":"2026-07-05T08:27:46.372290+00:00","updated_at":"2026-07-05T08:27:46.372290+00:00"}