{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Y7FBCUDMN44B4YX772HLQFNLQC","short_pith_number":"pith:Y7FBCUDM","schema_version":"1.0","canonical_sha256":"c7ca11506c6f381e62fffe8eb815ab80948316057d4f214d0ffb9c3a908afa26","source":{"kind":"arxiv","id":"2307.14539","version":2},"attestation_state":"computed","paper":{"title":"Jailbreak in pieces: Compositional Adversarial Attacks on Multi-Modal Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Erfan Shayegani, Nael Abu-Ghazaleh, Yue Dong","submitted_at":"2023-07-26T23:11:15Z","abstract_excerpt":"We introduce new jailbreak attacks on vision language models (VLMs), which use aligned LLMs and are resilient to text-only jailbreak attacks. Specifically, we develop cross-modality attacks on alignment where we pair adversarial images going through the vision encoder with textual prompts to break the alignment of the language model. Our attacks employ a novel compositional strategy that combines an image, adversarially targeted towards toxic embeddings, with generic prompts to accomplish the jailbreak. Thus, the LLM draws the context to answer the generic prompt from the adversarial image. Th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.14539","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2023-07-26T23:11:15Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8252ebc7a01dc4db277837a34671e7341de2863d5a81e66b4cea88ca97ea8172","abstract_canon_sha256":"eefd1597a15aef49b72fac7acf03deb50d73e32cb8118ffae4ff8192e0f23c10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:34.099632Z","signature_b64":"elgIl5Nzw8gMY8DTvKx7GbkjFfbWfYrtHo4TP9IWwCA2FuC7RpQVmdXFLp8PFi5QAYteHzbni3v3kVuOWqvFCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7ca11506c6f381e62fffe8eb815ab80948316057d4f214d0ffb9c3a908afa26","last_reissued_at":"2026-07-05T06:59:34.099246Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:34.099246Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Jailbreak in pieces: Compositional Adversarial Attacks on Multi-Modal Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Erfan Shayegani, Nael Abu-Ghazaleh, Yue Dong","submitted_at":"2023-07-26T23:11:15Z","abstract_excerpt":"We introduce new jailbreak attacks on vision language models (VLMs), which use aligned LLMs and are resilient to text-only jailbreak attacks. Specifically, we develop cross-modality attacks on alignment where we pair adversarial images going through the vision encoder with textual prompts to break the alignment of the language model. Our attacks employ a novel compositional strategy that combines an image, adversarially targeted towards toxic embeddings, with generic prompts to accomplish the jailbreak. Thus, the LLM draws the context to answer the generic prompt from the adversarial image. Th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.14539","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.14539/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.14539","created_at":"2026-07-05T06:59:34.099301+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.14539v2","created_at":"2026-07-05T06:59:34.099301+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.14539","created_at":"2026-07-05T06:59:34.099301+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7FBCUDMN44B","created_at":"2026-07-05T06:59:34.099301+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7FBCUDMN44B4YX7","created_at":"2026-07-05T06:59:34.099301+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7FBCUDM","created_at":"2026-07-05T06:59:34.099301+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07706","citing_title":"MLingualFC: Evaluating Jailbreak Vulnerabilities in Multilingual Vision-Language Models","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03793","citing_title":"Exploring Adversarial Robustness and Safety Alignment in Multilingual Multi-Modal Large Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25534","citing_title":"StructBreak: Structural Cognitive Overload-Induced Safety Failures in MLLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26566","citing_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14113","citing_title":"Adversarial Hubness in Multi-Modal Retrieval","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17128","citing_title":"New Wide-Net-Casting Jailbreak Attacks Risk Large Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21893","citing_title":"Breaking the Illusion: Consensus-Based Generative Mitigation of Adversarial Illusions in Multi-Modal Embeddings","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13213","citing_title":"Hierarchical Attacks for Multi-Modal Multi-Agent Reasoning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11716","citing_title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01449","citing_title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07490","citing_title":"Cross-Modal Backdoors in Multimodal Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07250","citing_title":"Hard to Read, Easy to Jailbreak: How Visual Degradation Bypasses MLLM Safety Alignment","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18803","citing_title":"LLM-as-Judge Framework for Evaluating Tone-Induced Hallucination in Vision-Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16966","citing_title":"Visual Inception: Compromising Long-term Planning in Agentic Recommenders via Multimodal Memory Poisoning","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC","json":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC.json","graph_json":"https://pith.science/api/pith-number/Y7FBCUDMN44B4YX772HLQFNLQC/graph.json","events_json":"https://pith.science/api/pith-number/Y7FBCUDMN44B4YX772HLQFNLQC/events.json","paper":"https://pith.science/paper/Y7FBCUDM"},"agent_actions":{"view_html":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC","download_json":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC.json","view_paper":"https://pith.science/paper/Y7FBCUDM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.14539&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7FBCUDMN44B4YX772HLQFNLQC/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7FBCUDMN44B4YX772HLQFNLQC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC/action/storage_attestation","attest_author":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC/action/author_attestation","sign_citation":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC/action/citation_signature","submit_replication":"https://pith.science/pith/Y7FBCUDMN44B4YX772HLQFNLQC/action/replication_record"}},"created_at":"2026-07-05T06:59:34.099301+00:00","updated_at":"2026-07-05T06:59:34.099301+00:00"}