{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PLFCJMHEKDQ6E4O5UNAGG67XJV","short_pith_number":"pith:PLFCJMHE","schema_version":"1.0","canonical_sha256":"7aca24b0e450e1e271dda340637bf74d6db94f4381b16ae88813652ad85bc699","source":{"kind":"arxiv","id":"2304.10436","version":1},"attestation_state":"computed","paper":{"title":"Safety Assessment of Chinese Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Sun, Jiale Cheng, Jiawen Deng, Minlie Huang, Zhexin Zhang","submitted_at":"2023-04-20T16:27:35Z","abstract_excerpt":"With the rapid popularity of large language models such as ChatGPT and GPT-4, a growing amount of attention is paid to their safety concerns. These models may generate insulting and discriminatory content, reflect incorrect social values, and may be used for malicious purposes such as fraud and dissemination of misleading information. Evaluating and enhancing their safety is particularly essential for the wide application of large language models (LLMs). To further promote the safe deployment of LLMs, we develop a Chinese LLM safety assessment benchmark. Our benchmark explores the comprehensiv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.10436","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-20T16:27:35Z","cross_cats_sorted":[],"title_canon_sha256":"bc35f214f3fb002528793a87871edd4ea540e4424e8df168ad744ea741a90536","abstract_canon_sha256":"1be44324059065dc7aec9df50d7a2343c5838fd514c67162fb09c2ea3595033c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:02:58.001261Z","signature_b64":"MQ5I+IcnZ9oHtqc+Fj7t5xzimscyqYVBasuzJKQici8fHYDIEUexN1Hr/S9FfdFrBEvdhO1yuoVePg8NQPbyAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7aca24b0e450e1e271dda340637bf74d6db94f4381b16ae88813652ad85bc699","last_reissued_at":"2026-07-05T06:02:58.000870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:02:58.000870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safety Assessment of Chinese Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Sun, Jiale Cheng, Jiawen Deng, Minlie Huang, Zhexin Zhang","submitted_at":"2023-04-20T16:27:35Z","abstract_excerpt":"With the rapid popularity of large language models such as ChatGPT and GPT-4, a growing amount of attention is paid to their safety concerns. These models may generate insulting and discriminatory content, reflect incorrect social values, and may be used for malicious purposes such as fraud and dissemination of misleading information. Evaluating and enhancing their safety is particularly essential for the wide application of large language models (LLMs). To further promote the safe deployment of LLMs, we develop a Chinese LLM safety assessment benchmark. Our benchmark explores the comprehensiv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.10436","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.10436/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.10436","created_at":"2026-07-05T06:02:58.000927+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.10436v1","created_at":"2026-07-05T06:02:58.000927+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.10436","created_at":"2026-07-05T06:02:58.000927+00:00"},{"alias_kind":"pith_short_12","alias_value":"PLFCJMHEKDQ6","created_at":"2026-07-05T06:02:58.000927+00:00"},{"alias_kind":"pith_short_16","alias_value":"PLFCJMHEKDQ6E4O5","created_at":"2026-07-05T06:02:58.000927+00:00"},{"alias_kind":"pith_short_8","alias_value":"PLFCJMHE","created_at":"2026-07-05T06:02:58.000927+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09178","citing_title":"Culturally-Adapted Red-Teaming Across East and Southeast Asian Contexts: A Methodological and Comparative Analysis","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24414","citing_title":"JT-SAFE-V2: Safety-by-Design Foundation Model with World-Context Data","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26954","citing_title":"AlbanianLLMSafety: A Safety Evaluation Dataset for Large Language Models in Albanian","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29667","citing_title":"Beyond English and Evasion: A Human-Annotated Multi-Domain Benchmark for High-Stakes LLM Safety Evaluation in Chinese","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22258","citing_title":"Harder to Defend: Towards Chinese Toxicity Attacks via Implicit Enhancement and Obfuscation Rewriting","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00432","citing_title":"Does Math Reasoning Improve General LLM Capabilities? Understanding Transferability of LLM Reasoning","ref_index":226,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2511.10287","citing_title":"OutSafe-Bench: A Benchmark for Multimodal Offensive Content Detection in Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2406.18495","citing_title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10253","citing_title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02954","citing_title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2405.04434","citing_title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06552","citing_title":"To Lie or Not to Lie? Investigating The Biased Spread of Global Lies by LLMs","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14548","citing_title":"VoxSafeBench: Not Just What Is Said, but Who, How, and Where","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16542","citing_title":"TWGuard: A Case Study of LLM Safety Guardrails for Localized Linguistic Contexts","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV","json":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV.json","graph_json":"https://pith.science/api/pith-number/PLFCJMHEKDQ6E4O5UNAGG67XJV/graph.json","events_json":"https://pith.science/api/pith-number/PLFCJMHEKDQ6E4O5UNAGG67XJV/events.json","paper":"https://pith.science/paper/PLFCJMHE"},"agent_actions":{"view_html":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV","download_json":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV.json","view_paper":"https://pith.science/paper/PLFCJMHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.10436&json=true","fetch_graph":"https://pith.science/api/pith-number/PLFCJMHEKDQ6E4O5UNAGG67XJV/graph.json","fetch_events":"https://pith.science/api/pith-number/PLFCJMHEKDQ6E4O5UNAGG67XJV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV/action/storage_attestation","attest_author":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV/action/author_attestation","sign_citation":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV/action/citation_signature","submit_replication":"https://pith.science/pith/PLFCJMHEKDQ6E4O5UNAGG67XJV/action/replication_record"}},"created_at":"2026-07-05T06:02:58.000927+00:00","updated_at":"2026-07-05T06:02:58.000927+00:00"}