{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LCSULM4LHXNXJLCWZHGHOT73K2","short_pith_number":"pith:LCSULM4L","schema_version":"1.0","canonical_sha256":"58a545b38b3ddb74ac56c9cc774ffb56980c44e5e446f7604c92b4d8c9d88c51","source":{"kind":"arxiv","id":"2311.17600","version":5},"attestation_state":"computed","paper":{"title":"MM-SafetyBench: A Benchmark for Safety Evaluation of Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chao Yang, Jindong Gu, Xin Liu, Yichen Zhu, Yunshi Lan, Yu Qiao","submitted_at":"2023-11-29T12:49:45Z","abstract_excerpt":"The security concerns surrounding Large Language Models (LLMs) have been extensively explored, yet the safety of Multimodal Large Language Models (MLLMs) remains understudied. In this paper, we observe that Multimodal Large Language Models (MLLMs) can be easily compromised by query-relevant images, as if the text query itself were malicious. To address this, we introduce MM-SafetyBench, a comprehensive framework designed for conducting safety-critical evaluations of MLLMs against such image-based manipulations. We have compiled a dataset comprising 13 scenarios, resulting in a total of 5,040 t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17600","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-29T12:49:45Z","cross_cats_sorted":[],"title_canon_sha256":"51cf67ce4d23fb915863f9e2fa77d24ae53eef7d059939d085561eb7b0094792","abstract_canon_sha256":"8a52c3194928bcb0b7ed9090d72fef4b5cbac51857ac5741e7d5802a5947754b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:08.629463Z","signature_b64":"TTgjb95/CMsEYOyvzp740Qf5RP/iv3fWY5cPwhub3epMdcrFn4f19zc+NHen9l+JA5FV5ZteJoVz4kS36mH5Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58a545b38b3ddb74ac56c9cc774ffb56980c44e5e446f7604c92b4d8c9d88c51","last_reissued_at":"2026-07-05T08:34:08.628981Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:08.628981Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MM-SafetyBench: A Benchmark for Safety Evaluation of Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chao Yang, Jindong Gu, Xin Liu, Yichen Zhu, Yunshi Lan, Yu Qiao","submitted_at":"2023-11-29T12:49:45Z","abstract_excerpt":"The security concerns surrounding Large Language Models (LLMs) have been extensively explored, yet the safety of Multimodal Large Language Models (MLLMs) remains understudied. In this paper, we observe that Multimodal Large Language Models (MLLMs) can be easily compromised by query-relevant images, as if the text query itself were malicious. To address this, we introduce MM-SafetyBench, a comprehensive framework designed for conducting safety-critical evaluations of MLLMs against such image-based manipulations. We have compiled a dataset comprising 13 scenarios, resulting in a total of 5,040 t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17600","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17600/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17600","created_at":"2026-07-05T08:34:08.629040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17600v5","created_at":"2026-07-05T08:34:08.629040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17600","created_at":"2026-07-05T08:34:08.629040+00:00"},{"alias_kind":"pith_short_12","alias_value":"LCSULM4LHXNX","created_at":"2026-07-05T08:34:08.629040+00:00"},{"alias_kind":"pith_short_16","alias_value":"LCSULM4LHXNXJLCW","created_at":"2026-07-05T08:34:08.629040+00:00"},{"alias_kind":"pith_short_8","alias_value":"LCSULM4L","created_at":"2026-07-05T08:34:08.629040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05910","citing_title":"PolicyShiftGuard: Benchmarking and Improving Policy-Adaptive Image Guardrails","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09125","citing_title":"Unveiling Privacy Risks in Multi-modal Large Language Models: Task-specific Vulnerabilities and Mitigation Challenges","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30049","citing_title":"Robust and Generalizable Safety Steering for Text-to-Image Diffusion Transformers","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26566","citing_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2503.06223","citing_title":"RedDiffuser: Auditing Multimodal Safety Failures in Vision-Language Models via Reinforced Diffusion","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11716","citing_title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01449","citing_title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05900","citing_title":"AICA-Bench: Holistically Examining the Capabilities of VLMs in Affective Image Content Analysis","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13803","citing_title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2","json":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2.json","graph_json":"https://pith.science/api/pith-number/LCSULM4LHXNXJLCWZHGHOT73K2/graph.json","events_json":"https://pith.science/api/pith-number/LCSULM4LHXNXJLCWZHGHOT73K2/events.json","paper":"https://pith.science/paper/LCSULM4L"},"agent_actions":{"view_html":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2","download_json":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2.json","view_paper":"https://pith.science/paper/LCSULM4L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17600&json=true","fetch_graph":"https://pith.science/api/pith-number/LCSULM4LHXNXJLCWZHGHOT73K2/graph.json","fetch_events":"https://pith.science/api/pith-number/LCSULM4LHXNXJLCWZHGHOT73K2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2/action/storage_attestation","attest_author":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2/action/author_attestation","sign_citation":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2/action/citation_signature","submit_replication":"https://pith.science/pith/LCSULM4LHXNXJLCWZHGHOT73K2/action/replication_record"}},"created_at":"2026-07-05T08:34:08.629040+00:00","updated_at":"2026-07-05T08:34:08.629040+00:00"}