{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:33EWEE64LLPUHJ3PIPUF77ODEM","short_pith_number":"pith:33EWEE64","schema_version":"1.0","canonical_sha256":"dec96213dc5adf43a76f43e85ffdc32327c40011ba1be9c70c9caa3729423a00","source":{"kind":"arxiv","id":"2403.09792","version":3},"attestation_state":"computed","paper":{"title":"Images are Achilles' Heel of Alignment: Exploiting Visual Vulnerabilities for Jailbreaking Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Hangyu Guo, Ji-Rong Wen, Kun Zhou, Wayne Xin Zhao, Yifan Li","submitted_at":"2024-03-14T18:24:55Z","abstract_excerpt":"In this paper, we study the harmlessness alignment problem of multimodal large language models (MLLMs). We conduct a systematic empirical analysis of the harmlessness performance of representative MLLMs and reveal that the image input poses the alignment vulnerability of MLLMs. Inspired by this, we propose a novel jailbreak method named HADES, which hides and amplifies the harmfulness of the malicious intent within the text input, using meticulously crafted images. Experimental results show that HADES can effectively jailbreak existing MLLMs, which achieves an average Attack Success Rate (ASR)"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.09792","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-14T18:24:55Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"9d7414148a546713f8e3766360d0960e24a8ef17b63ddf3fbe5ecfd0f29c9a43","abstract_canon_sha256":"e3cc4e45bd2717bb8756f986f809d0292cf39330ccbe160eb0a61f6cc14b7209"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:44.862405Z","signature_b64":"1J6dOc0P3PBiIVIb7K7n1CFttNImWFe8OAGf9iRFrASFhBFmgs83LQYYQJHOn21ojazzmBd9cLUyUIcpU/KvDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dec96213dc5adf43a76f43e85ffdc32327c40011ba1be9c70c9caa3729423a00","last_reissued_at":"2026-07-05T09:59:44.861908Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:44.861908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Images are Achilles' Heel of Alignment: Exploiting Visual Vulnerabilities for Jailbreaking Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Hangyu Guo, Ji-Rong Wen, Kun Zhou, Wayne Xin Zhao, Yifan Li","submitted_at":"2024-03-14T18:24:55Z","abstract_excerpt":"In this paper, we study the harmlessness alignment problem of multimodal large language models (MLLMs). We conduct a systematic empirical analysis of the harmlessness performance of representative MLLMs and reveal that the image input poses the alignment vulnerability of MLLMs. Inspired by this, we propose a novel jailbreak method named HADES, which hides and amplifies the harmfulness of the malicious intent within the text input, using meticulously crafted images. Experimental results show that HADES can effectively jailbreak existing MLLMs, which achieves an average Attack Success Rate (ASR)"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.09792","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.09792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.09792","created_at":"2026-07-05T09:59:44.861967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.09792v3","created_at":"2026-07-05T09:59:44.861967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.09792","created_at":"2026-07-05T09:59:44.861967+00:00"},{"alias_kind":"pith_short_12","alias_value":"33EWEE64LLPU","created_at":"2026-07-05T09:59:44.861967+00:00"},{"alias_kind":"pith_short_16","alias_value":"33EWEE64LLPUHJ3P","created_at":"2026-07-05T09:59:44.861967+00:00"},{"alias_kind":"pith_short_8","alias_value":"33EWEE64","created_at":"2026-07-05T09:59:44.861967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14750","citing_title":"EVA: Editing for Versatile Alignment against Jailbreaks","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07595","citing_title":"VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26566","citing_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10582","citing_title":"Guaranteed Jailbreaking Defense via Disrupt-and-Rectify Smoothing","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01449","citing_title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06247","citing_title":"SALLIE: Safeguarding Against Latent Language & Image Exploits","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13803","citing_title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM","json":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM.json","graph_json":"https://pith.science/api/pith-number/33EWEE64LLPUHJ3PIPUF77ODEM/graph.json","events_json":"https://pith.science/api/pith-number/33EWEE64LLPUHJ3PIPUF77ODEM/events.json","paper":"https://pith.science/paper/33EWEE64"},"agent_actions":{"view_html":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM","download_json":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM.json","view_paper":"https://pith.science/paper/33EWEE64","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.09792&json=true","fetch_graph":"https://pith.science/api/pith-number/33EWEE64LLPUHJ3PIPUF77ODEM/graph.json","fetch_events":"https://pith.science/api/pith-number/33EWEE64LLPUHJ3PIPUF77ODEM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM/action/storage_attestation","attest_author":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM/action/author_attestation","sign_citation":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM/action/citation_signature","submit_replication":"https://pith.science/pith/33EWEE64LLPUHJ3PIPUF77ODEM/action/replication_record"}},"created_at":"2026-07-05T09:59:44.861967+00:00","updated_at":"2026-07-05T09:59:44.861967+00:00"}