{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ADB2YYD2VHCEMS3ZJ7XZFQWH6I","short_pith_number":"pith:ADB2YYD2","schema_version":"1.0","canonical_sha256":"00c3ac607aa9c4464b794fef92c2c7f208d80b81fe5bbd1ceab70507db76c7c2","source":{"kind":"arxiv","id":"2310.06474","version":3},"attestation_state":"computed","paper":{"title":"Multilingual Jailbreak Challenges in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lidong Bing, Sinno Jialin Pan, Wenxuan Zhang, Yue Deng","submitted_at":"2023-10-10T09:44:06Z","abstract_excerpt":"While large language models (LLMs) exhibit remarkable capabilities across a wide range of tasks, they pose potential safety concerns, such as the ``jailbreak'' problem, wherein malicious instructions can manipulate LLMs to exhibit undesirable behavior. Although several preventive measures have been developed to mitigate the potential risks associated with LLMs, they have primarily focused on English. In this study, we reveal the presence of multilingual jailbreak challenges within LLMs and consider two potential risky scenarios: unintentional and intentional. The unintentional scenario involve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.06474","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-10T09:44:06Z","cross_cats_sorted":[],"title_canon_sha256":"7a5ef5be9f1f1e5b6b4fe5f7a51787497f61330354820e6b5bcd9eba60f81089","abstract_canon_sha256":"7b956b4ce046d674c48d7546c35a79c49284f062fe3215af4546f94305450d1b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:33.554819Z","signature_b64":"MijexDnvQo6rSXSi30w9AvuxegxVyKUXrzQzLVb1R7CgYvviNI+4igreR5R1cLgoKih4lv9JQNx6SNy8/WR/Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00c3ac607aa9c4464b794fef92c2c7f208d80b81fe5bbd1ceab70507db76c7c2","last_reissued_at":"2026-07-05T07:51:33.554226Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:33.554226Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multilingual Jailbreak Challenges in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lidong Bing, Sinno Jialin Pan, Wenxuan Zhang, Yue Deng","submitted_at":"2023-10-10T09:44:06Z","abstract_excerpt":"While large language models (LLMs) exhibit remarkable capabilities across a wide range of tasks, they pose potential safety concerns, such as the ``jailbreak'' problem, wherein malicious instructions can manipulate LLMs to exhibit undesirable behavior. Although several preventive measures have been developed to mitigate the potential risks associated with LLMs, they have primarily focused on English. In this study, we reveal the presence of multilingual jailbreak challenges within LLMs and consider two potential risky scenarios: unintentional and intentional. The unintentional scenario involve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.06474","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.06474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.06474","created_at":"2026-07-05T07:51:33.554284+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.06474v3","created_at":"2026-07-05T07:51:33.554284+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.06474","created_at":"2026-07-05T07:51:33.554284+00:00"},{"alias_kind":"pith_short_12","alias_value":"ADB2YYD2VHCE","created_at":"2026-07-05T07:51:33.554284+00:00"},{"alias_kind":"pith_short_16","alias_value":"ADB2YYD2VHCEMS3Z","created_at":"2026-07-05T07:51:33.554284+00:00"},{"alias_kind":"pith_short_8","alias_value":"ADB2YYD2","created_at":"2026-07-05T07:51:33.554284+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07833","citing_title":"Beyond Pass/Fail: Using Process Mining to Understand How LLMs Resist (and Fail) Red Team Attacks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05523","citing_title":"CHASE: Adversarial Red-Blue Teaming for Improving LLM Safety using Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29602","citing_title":"An Empirical Evaluation of Prompt Injection Vulnerabilities in Large Language Models Across Multilingual and Obfuscated Attack Scenarios","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25420","citing_title":"SomaliBench Eval: Measuring English-to-Somali Refusal Gaps in Open-Weight Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29659","citing_title":"Opir: Efficient Multi-Task Safety Classification for Toxicity, Jailbreaks, Hate Speech, and Harmful Content","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01196","citing_title":"Low-Resource Safety Failures Are Action Failures, Not Representation Failures","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23157","citing_title":"Same Model, Different Weakness: How Language and Modality Reshape the Jailbreak Attack Surface in Frontier MLLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01941","citing_title":"Semantic Integrity Matters: Benchmarking and Preserving High-Density Reasoning in KV Cache Compression","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16471","citing_title":"From AI-Generated Content to Agentic Action: Security and Safety Threats in Generative AI","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17173","citing_title":"Why Do Safety Guardrails Degrade Across Languages?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18239","citing_title":"Multilingual jailbreaking of LLMs using low-resource languages","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19328","citing_title":"RoboJailBench: Benchmarking Adversarial Attacks and Defenses in Embodied Robotic Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10073","citing_title":"SecureWebArena: A Holistic Security Evaluation Benchmark for LVLM-based Web Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01833","citing_title":"Great, Now Write an Article About That: The Crescendo Multi-Turn LLM Jailbreak Attack","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11157","citing_title":"Response-Based Knowledge Distillation for Multilingual Jailbreak Prevention Unwittingly Compromises Safety","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14152","citing_title":"ROK-FORTRESS: Measuring the Effect of Geopolitical Transcreation for National Security and Public Safety","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14152","citing_title":"ROK-FORTRESS: Measuring the Effect of Geopolitical Transcreation for National Security and Public Safety","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01473","citing_title":"SelfGrader: LLM Jailbreak Detection via Anchored Token-Level Logits","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05662","citing_title":"XL-SafetyBench: A Country-Grounded Cross-Cultural Benchmark for LLM Safety and Cultural Sensitivity","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20720","citing_title":"COMPASS: COntinual Multilingual PEFT with Adaptive Semantic Sampling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07727","citing_title":"TrajGuard: Streaming Hidden-state Trajectory Detection for Decoding-time Jailbreak Defense","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I","json":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I.json","graph_json":"https://pith.science/api/pith-number/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/graph.json","events_json":"https://pith.science/api/pith-number/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/events.json","paper":"https://pith.science/paper/ADB2YYD2"},"agent_actions":{"view_html":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I","download_json":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I.json","view_paper":"https://pith.science/paper/ADB2YYD2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.06474&json=true","fetch_graph":"https://pith.science/api/pith-number/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/graph.json","fetch_events":"https://pith.science/api/pith-number/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/action/storage_attestation","attest_author":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/action/author_attestation","sign_citation":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/action/citation_signature","submit_replication":"https://pith.science/pith/ADB2YYD2VHCEMS3ZJ7XZFQWH6I/action/replication_record"}},"created_at":"2026-07-05T07:51:33.554284+00:00","updated_at":"2026-07-05T07:51:33.554284+00:00"}