{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FNA6XVC5675OQ3HXT4SJ363LM2","short_pith_number":"pith:FNA6XVC5","schema_version":"1.0","canonical_sha256":"2b41ebd45df7fae86cf79f249dfb6b6690ca87c8d1e246658f8b822058d29dca","source":{"kind":"arxiv","id":"2309.07045","version":2},"attestation_state":"computed","paper":{"title":"SafetyBench: Evaluating the Safety of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chong Long, Jie Tang, Leqi Lei, Lindong Wu, Minlie Huang, Rui Sun, Xiao Liu, Xuanyu Lei, Yongkang Huang, Zhexin Zhang","submitted_at":"2023-09-13T15:56:50Z","abstract_excerpt":"With the rapid development of Large Language Models (LLMs), increasing attention has been paid to their safety concerns. Consequently, evaluating the safety of LLMs has become an essential task for facilitating the broad applications of LLMs. Nevertheless, the absence of comprehensive safety evaluation benchmarks poses a significant impediment to effectively assess and enhance the safety of LLMs. In this work, we present SafetyBench, a comprehensive benchmark for evaluating the safety of LLMs, which comprises 11,435 diverse multiple choice questions spanning across 7 distinct categories of saf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.07045","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-13T15:56:50Z","cross_cats_sorted":[],"title_canon_sha256":"0f5e4000ce009452d6ad60460b1fadd8ba6441bb5ad640e9e47296914bd79b15","abstract_canon_sha256":"9d35d37da79dc781c2abfae62ea3b384f7a57ab2f6ea13539a9d3c42fbda181e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:36.836470Z","signature_b64":"vWo6soRh6tGiVvFG2I8Y++r7WwSekFtN6Yu65ylcOu5xjovisSzo9DOoGgI/fNqK+Ql21ixT32OZFIxiMaOaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b41ebd45df7fae86cf79f249dfb6b6690ca87c8d1e246658f8b822058d29dca","last_reissued_at":"2026-07-05T08:35:36.835986Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:36.835986Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafetyBench: Evaluating the Safety of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chong Long, Jie Tang, Leqi Lei, Lindong Wu, Minlie Huang, Rui Sun, Xiao Liu, Xuanyu Lei, Yongkang Huang, Zhexin Zhang","submitted_at":"2023-09-13T15:56:50Z","abstract_excerpt":"With the rapid development of Large Language Models (LLMs), increasing attention has been paid to their safety concerns. Consequently, evaluating the safety of LLMs has become an essential task for facilitating the broad applications of LLMs. Nevertheless, the absence of comprehensive safety evaluation benchmarks poses a significant impediment to effectively assess and enhance the safety of LLMs. In this work, we present SafetyBench, a comprehensive benchmark for evaluating the safety of LLMs, which comprises 11,435 diverse multiple choice questions spanning across 7 distinct categories of saf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.07045","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.07045/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.07045","created_at":"2026-07-05T08:35:36.836044+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.07045v2","created_at":"2026-07-05T08:35:36.836044+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.07045","created_at":"2026-07-05T08:35:36.836044+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNA6XVC5675O","created_at":"2026-07-05T08:35:36.836044+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNA6XVC5675OQ3HX","created_at":"2026-07-05T08:35:36.836044+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNA6XVC5","created_at":"2026-07-05T08:35:36.836044+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00913","citing_title":"Two AI Metrics Diverged: Will it Make All the Difference?","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20626","citing_title":"Efficient Safety Benchmarking via Item Response Theory","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27632","citing_title":"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21110","citing_title":"Beyond Context: Large Language Models' Failure to Grasp Users' Intent","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02713","citing_title":"Breakdowns in Conversational AI: Interactional Failures in Emotionally and Ethically Sensitive Contexts","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10639","citing_title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24074","citing_title":"How Sensitive Are Safety Benchmarks to Judge Configuration Choices?","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06471","citing_title":"GLM-4.5: Agentic, Reasoning, and Coding (ARC) Foundation Models","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2406.12793","citing_title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14548","citing_title":"VoxSafeBench: Not Just What Is Said, but Who, How, and Where","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16659","citing_title":"Benign Fine-Tuning Breaks Safety Alignment in Audio LLMs","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06652","citing_title":"When No Benchmark Exists: Validating Comparative LLM Safety Scoring Without Ground-Truth Labels","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2","json":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2.json","graph_json":"https://pith.science/api/pith-number/FNA6XVC5675OQ3HXT4SJ363LM2/graph.json","events_json":"https://pith.science/api/pith-number/FNA6XVC5675OQ3HXT4SJ363LM2/events.json","paper":"https://pith.science/paper/FNA6XVC5"},"agent_actions":{"view_html":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2","download_json":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2.json","view_paper":"https://pith.science/paper/FNA6XVC5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.07045&json=true","fetch_graph":"https://pith.science/api/pith-number/FNA6XVC5675OQ3HXT4SJ363LM2/graph.json","fetch_events":"https://pith.science/api/pith-number/FNA6XVC5675OQ3HXT4SJ363LM2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2/action/storage_attestation","attest_author":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2/action/author_attestation","sign_citation":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2/action/citation_signature","submit_replication":"https://pith.science/pith/FNA6XVC5675OQ3HXT4SJ363LM2/action/replication_record"}},"created_at":"2026-07-05T08:35:36.836044+00:00","updated_at":"2026-07-05T08:35:36.836044+00:00"}