{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IQSNFS6UUYZHB5EIAMXJKF6YSJ","short_pith_number":"pith:IQSNFS6U","schema_version":"1.0","canonical_sha256":"4424d2cbd4a63270f488032e9517d8926390d123cbd4feff70a3f9f0fee8dd72","source":{"kind":"arxiv","id":"2310.00905","version":2},"attestation_state":"computed","paper":{"title":"All Languages Matter: On the Multilingual Safety of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chang Chen, Jen-tse Huang, Michael R. Lyu, Wenxiang Jiao, Wenxuan Wang, Youliang Yuan, Zhaopeng Tu","submitted_at":"2023-10-02T05:23:34Z","abstract_excerpt":"Safety lies at the core of developing and deploying large language models (LLMs). However, previous safety benchmarks only concern the safety in one language, e.g. the majority language in the pretraining data such as English. In this work, we build the first multilingual safety benchmark for LLMs, XSafety, in response to the global deployment of LLMs in practice. XSafety covers 14 kinds of commonly used safety issues across 10 languages that span several language families. We utilize XSafety to empirically study the multilingual safety for 4 widely-used LLMs, including both close-API and open"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00905","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-02T05:23:34Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"225d11c4b2b96d02a529eb46b2dc3e64b12aa0744d17c16a1860e2090df235fc","abstract_canon_sha256":"a449ab218dbea169bb945469733719e678f5fc50a5b548595677ebf3b5f44a93"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:46.819801Z","signature_b64":"bWO0Uujsq51qdwxfC3i+pcvpRH4zHmaVSmqqO/OMCAWNT/Rbg8rQnnUH/81+rwdRYLdaJWirn7oi5p9EPaDZAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4424d2cbd4a63270f488032e9517d8926390d123cbd4feff70a3f9f0fee8dd72","last_reissued_at":"2026-07-05T08:34:46.819289Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:46.819289Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"All Languages Matter: On the Multilingual Safety of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chang Chen, Jen-tse Huang, Michael R. Lyu, Wenxiang Jiao, Wenxuan Wang, Youliang Yuan, Zhaopeng Tu","submitted_at":"2023-10-02T05:23:34Z","abstract_excerpt":"Safety lies at the core of developing and deploying large language models (LLMs). However, previous safety benchmarks only concern the safety in one language, e.g. the majority language in the pretraining data such as English. In this work, we build the first multilingual safety benchmark for LLMs, XSafety, in response to the global deployment of LLMs in practice. XSafety covers 14 kinds of commonly used safety issues across 10 languages that span several language families. We utilize XSafety to empirically study the multilingual safety for 4 widely-used LLMs, including both close-API and open"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00905","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00905/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00905","created_at":"2026-07-05T08:34:46.819360+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00905v2","created_at":"2026-07-05T08:34:46.819360+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00905","created_at":"2026-07-05T08:34:46.819360+00:00"},{"alias_kind":"pith_short_12","alias_value":"IQSNFS6UUYZH","created_at":"2026-07-05T08:34:46.819360+00:00"},{"alias_kind":"pith_short_16","alias_value":"IQSNFS6UUYZHB5EI","created_at":"2026-07-05T08:34:46.819360+00:00"},{"alias_kind":"pith_short_8","alias_value":"IQSNFS6U","created_at":"2026-07-05T08:34:46.819360+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07535","citing_title":"Multilingual Refusal Alignment for Safer Large Language Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09178","citing_title":"Culturally-Adapted Red-Teaming Across East and Southeast Asian Contexts: A Methodological and Comparative Analysis","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28733","citing_title":"Agentic Abstention: Do Agents Know When to Stop Instead of Act?","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00557","citing_title":"Learning to Ask: When LLM Agents Meet Unclear Instruction","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12696","citing_title":"UbuntuGuard: A Culturally-Grounded Policy Benchmark for Equitable AI Safety in African Languages","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17173","citing_title":"Why Do Safety Guardrails Degrade Across Languages?","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11157","citing_title":"Response-Based Knowledge Distillation for Multilingual Jailbreak Prevention Unwittingly Compromises Safety","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14152","citing_title":"ROK-FORTRESS: Measuring the Effect of Geopolitical Transcreation for National Security and Public Safety","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11026","citing_title":"AgentShield: Deception-based Compromise Detection for Tool-using LLM Agents","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02971","citing_title":"Multilingual Safety Alignment via Self-Distillation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02971","citing_title":"Multilingual Safety Alignment via Self-Distillation","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ","json":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ.json","graph_json":"https://pith.science/api/pith-number/IQSNFS6UUYZHB5EIAMXJKF6YSJ/graph.json","events_json":"https://pith.science/api/pith-number/IQSNFS6UUYZHB5EIAMXJKF6YSJ/events.json","paper":"https://pith.science/paper/IQSNFS6U"},"agent_actions":{"view_html":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ","download_json":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ.json","view_paper":"https://pith.science/paper/IQSNFS6U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00905&json=true","fetch_graph":"https://pith.science/api/pith-number/IQSNFS6UUYZHB5EIAMXJKF6YSJ/graph.json","fetch_events":"https://pith.science/api/pith-number/IQSNFS6UUYZHB5EIAMXJKF6YSJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ/action/storage_attestation","attest_author":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ/action/author_attestation","sign_citation":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ/action/citation_signature","submit_replication":"https://pith.science/pith/IQSNFS6UUYZHB5EIAMXJKF6YSJ/action/replication_record"}},"created_at":"2026-07-05T08:34:46.819360+00:00","updated_at":"2026-07-05T08:34:46.819360+00:00"}