{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MJMQSQE5RPCWQMP42YESCZZE4P","short_pith_number":"pith:MJMQSQE5","schema_version":"1.0","canonical_sha256":"625909409d8bc56831fcd609216724e3ce872b3c1b8bd37e7ef889628fa3f998","source":{"kind":"arxiv","id":"2405.20773","version":2},"attestation_state":"computed","paper":{"title":"Visual-RolePlay: Universal Jailbreak Attack on MultiModal Large Language Models via Role-playing Image Character","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Siyuan Ma, Weidi Luo, Xiaogeng Liu, Yu Wang","submitted_at":"2024-05-25T17:17:18Z","abstract_excerpt":"With the advent and widespread deployment of Multimodal Large Language Models (MLLMs), ensuring their safety has become increasingly critical. To achieve this objective, it requires us to proactively discover the vulnerability of MLLMs by exploring the attack methods. Thus, structure-based jailbreak attacks, where harmful semantic content is embedded within images, have been proposed to mislead the models. However, previous structure-based jailbreak methods mainly focus on transforming the format of malicious queries, such as converting harmful content into images through typography, which lac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20773","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-05-25T17:17:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ea6196bf1fcc96d336b3b18734bcfa5f4d65f5557fac82b555b339d832d559f6","abstract_canon_sha256":"5d36e2c8ff240619ce34a3886209d2aa1cb8c5d5861f4fa145b022dc99089942"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:33.846445Z","signature_b64":"jTWE6zAbwU4W23ZTCJb4N0RVDv9kA1RzaRHy1uUjEWRJkODyw/ZOpdxSGF3TnYqcJxlqHU5+O39I2W8dGMWwDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"625909409d8bc56831fcd609216724e3ce872b3c1b8bd37e7ef889628fa3f998","last_reissued_at":"2026-07-05T08:30:33.845933Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:33.845933Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual-RolePlay: Universal Jailbreak Attack on MultiModal Large Language Models via Role-playing Image Character","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Siyuan Ma, Weidi Luo, Xiaogeng Liu, Yu Wang","submitted_at":"2024-05-25T17:17:18Z","abstract_excerpt":"With the advent and widespread deployment of Multimodal Large Language Models (MLLMs), ensuring their safety has become increasingly critical. To achieve this objective, it requires us to proactively discover the vulnerability of MLLMs by exploring the attack methods. Thus, structure-based jailbreak attacks, where harmful semantic content is embedded within images, have been proposed to mislead the models. However, previous structure-based jailbreak methods mainly focus on transforming the format of malicious queries, such as converting harmful content into images through typography, which lac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20773","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20773/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20773","created_at":"2026-07-05T08:30:33.845999+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20773v2","created_at":"2026-07-05T08:30:33.845999+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20773","created_at":"2026-07-05T08:30:33.845999+00:00"},{"alias_kind":"pith_short_12","alias_value":"MJMQSQE5RPCW","created_at":"2026-07-05T08:30:33.845999+00:00"},{"alias_kind":"pith_short_16","alias_value":"MJMQSQE5RPCWQMP4","created_at":"2026-07-05T08:30:33.845999+00:00"},{"alias_kind":"pith_short_8","alias_value":"MJMQSQE5","created_at":"2026-07-05T08:30:33.845999+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07706","citing_title":"MLingualFC: Evaluating Jailbreak Vulnerabilities in Multilingual Vision-Language Models","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25194","citing_title":"Localization then Neutralization: Gradient-guided Token Suppression against Visual Prompt Injection Attack","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26566","citing_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":283,"is_internal_anchor":false},{"citing_arxiv_id":"2503.06223","citing_title":"RedDiffuser: Auditing Multimodal Safety Failures in Vision-Language Models via Reinforced Diffusion","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09443","citing_title":"Through the Lens of Character: Resolving Modality-Role Interference in Multimodal Role-Playing Agent","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08881","citing_title":"Targeted Interpretable Safety Neuron Enhancement for Multilingual Vision-Language Large Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06950","citing_title":"Making MLLMs Blind: Adversarial Smuggling Attacks in MLLM Content Moderation","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P","json":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P.json","graph_json":"https://pith.science/api/pith-number/MJMQSQE5RPCWQMP42YESCZZE4P/graph.json","events_json":"https://pith.science/api/pith-number/MJMQSQE5RPCWQMP42YESCZZE4P/events.json","paper":"https://pith.science/paper/MJMQSQE5"},"agent_actions":{"view_html":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P","download_json":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P.json","view_paper":"https://pith.science/paper/MJMQSQE5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20773&json=true","fetch_graph":"https://pith.science/api/pith-number/MJMQSQE5RPCWQMP42YESCZZE4P/graph.json","fetch_events":"https://pith.science/api/pith-number/MJMQSQE5RPCWQMP42YESCZZE4P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P/action/storage_attestation","attest_author":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P/action/author_attestation","sign_citation":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P/action/citation_signature","submit_replication":"https://pith.science/pith/MJMQSQE5RPCWQMP42YESCZZE4P/action/replication_record"}},"created_at":"2026-07-05T08:30:33.845999+00:00","updated_at":"2026-07-05T08:30:33.845999+00:00"}