{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AX4MZOWTKLYONWGAMIN3KK3SEI","short_pith_number":"pith:AX4MZOWT","schema_version":"1.0","canonical_sha256":"05f8ccbad352f0e6d8c0621bb52b72223de5314335d1bba1126f7866d44925f6","source":{"kind":"arxiv","id":"2404.03027","version":4},"attestation_state":"computed","paper":{"title":"JailBreakV: A Benchmark for Assessing the Robustness of MultiModal Large Language Models against Jailbreak Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Chaowei Xiao, Siyuan Ma, Weidi Luo, Xiaogeng Liu, Xiaoyu Guo","submitted_at":"2024-04-03T19:23:18Z","abstract_excerpt":"With the rapid advancements in Multimodal Large Language Models (MLLMs), securing these models against malicious inputs while aligning them with human values has emerged as a critical challenge. In this paper, we investigate an important and unexplored question of whether techniques that successfully jailbreak Large Language Models (LLMs) can be equally effective in jailbreaking MLLMs. To explore this issue, we introduce JailBreakV-28K, a pioneering benchmark designed to assess the transferability of LLM jailbreak techniques to MLLMs, thereby evaluating the robustness of MLLMs against diverse "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03027","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-04-03T19:23:18Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"69f41dfeede5dd9a4a872618aa9480beefe3ac3a22ef0e068b4ad79e7b228611","abstract_canon_sha256":"b4477e5f3796336d9ea0000cd74354f6ecb37967be089f9969cb8389c046bad3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:22.235567Z","signature_b64":"9sHM1DOng9Zyy/e3AR/DorgnSJdC5FDqVUM3rjapdZyu21QHaTFmdfHYtLcQ+crrlkHFQ6ElW74N7dI3/jbNCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"05f8ccbad352f0e6d8c0621bb52b72223de5314335d1bba1126f7866d44925f6","last_reissued_at":"2026-07-05T09:39:22.235105Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:22.235105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JailBreakV: A Benchmark for Assessing the Robustness of MultiModal Large Language Models against Jailbreak Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Chaowei Xiao, Siyuan Ma, Weidi Luo, Xiaogeng Liu, Xiaoyu Guo","submitted_at":"2024-04-03T19:23:18Z","abstract_excerpt":"With the rapid advancements in Multimodal Large Language Models (MLLMs), securing these models against malicious inputs while aligning them with human values has emerged as a critical challenge. In this paper, we investigate an important and unexplored question of whether techniques that successfully jailbreak Large Language Models (LLMs) can be equally effective in jailbreaking MLLMs. To explore this issue, we introduce JailBreakV-28K, a pioneering benchmark designed to assess the transferability of LLM jailbreak techniques to MLLMs, thereby evaluating the robustness of MLLMs against diverse "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03027","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03027/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03027","created_at":"2026-07-05T09:39:22.235158+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03027v4","created_at":"2026-07-05T09:39:22.235158+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03027","created_at":"2026-07-05T09:39:22.235158+00:00"},{"alias_kind":"pith_short_12","alias_value":"AX4MZOWTKLYO","created_at":"2026-07-05T09:39:22.235158+00:00"},{"alias_kind":"pith_short_16","alias_value":"AX4MZOWTKLYONWGA","created_at":"2026-07-05T09:39:22.235158+00:00"},{"alias_kind":"pith_short_8","alias_value":"AX4MZOWT","created_at":"2026-07-05T09:39:22.235158+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07907","citing_title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","ref_index":234,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24388","citing_title":"PHANTOM: A Large-Scale Dataset of Multimodal Adversarial Attacks for Vision-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20508","citing_title":"What Do Safety-Aligned LLMs Learn From Mixed Compliance Demonstrations?","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18632","citing_title":"ROBOSHACKLES: A Safety Dataset for Human-Injury Prevention in Embodied Foundation Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07335","citing_title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27932","citing_title":"When Think-with-Image Meets Safety: What Determines Multimodal Jailbreak Robustness?","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01481","citing_title":"SafeGen-Bench: Benchmarking Safety in Image-Conditioned Text-to-Video Generation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07706","citing_title":"MLingualFC: Evaluating Jailbreak Vulnerabilities in Multilingual Vision-Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2504.00446","citing_title":"Exposing the Ghost in the Transformer: Abnormal Detection for Large Language Models via Hidden State Forensics","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18868","citing_title":"DarkLLM: Learning Language-Driven Adversarial Attacks with Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21540","citing_title":"PRISM: Programmatic Reasoning with Image Sequence Manipulation for LVLM Jailbreaking","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20856","citing_title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21815","citing_title":"High-Entropy Tokens as Multimodal Failure Points in Vision-Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2603.21697","citing_title":"Structured Visual Narratives Undermine Safety Alignment in Multimodal Large Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05678","citing_title":"Chain of Risk: Safety Failures in Large Reasoning Models and Mitigation via Adaptive Multi-Principle Steering","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12374","citing_title":"Nemotron 3 Super: Open, Efficient Mixture-of-Experts Hybrid Mamba-Transformer Model for Agentic Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01687","citing_title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05498","citing_title":"JailWAM: Jailbreaking World Action Models in Robot Control","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04446","citing_title":"Misrouter: Exploiting Routing Mechanisms for Input-Only Attacks on Mixture-of-Experts LLMs","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI","json":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI.json","graph_json":"https://pith.science/api/pith-number/AX4MZOWTKLYONWGAMIN3KK3SEI/graph.json","events_json":"https://pith.science/api/pith-number/AX4MZOWTKLYONWGAMIN3KK3SEI/events.json","paper":"https://pith.science/paper/AX4MZOWT"},"agent_actions":{"view_html":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI","download_json":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI.json","view_paper":"https://pith.science/paper/AX4MZOWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03027&json=true","fetch_graph":"https://pith.science/api/pith-number/AX4MZOWTKLYONWGAMIN3KK3SEI/graph.json","fetch_events":"https://pith.science/api/pith-number/AX4MZOWTKLYONWGAMIN3KK3SEI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI/action/storage_attestation","attest_author":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI/action/author_attestation","sign_citation":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI/action/citation_signature","submit_replication":"https://pith.science/pith/AX4MZOWTKLYONWGAMIN3KK3SEI/action/replication_record"}},"created_at":"2026-07-05T09:39:22.235158+00:00","updated_at":"2026-07-05T09:39:22.235158+00:00"}