{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EKCWX66LPDCNHJOPP7MAIFWFKP","short_pith_number":"pith:EKCWX66L","schema_version":"1.0","canonical_sha256":"22856bfbcb78c4d3a5cf7fd80416c553d6e91f89f439a56c1256acdaec92dd42","source":{"kind":"arxiv","id":"2405.14191","version":4},"attestation_state":"computed","paper":{"title":"S-Eval: Towards Automated and Comprehensive Safety Evaluation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Dongxia Wang, Hui Xue, Jialuo Chen, Jinfeng Li, Jingyi Wang, Kui Ren, Longtao Huang, Wenhai Wang, Xiaofeng Mao, Xiaohan Yuan, Xiaoxia Liu, Yuefeng Chen","submitted_at":"2024-05-23T05:34:31Z","abstract_excerpt":"Generative large language models (LLMs) have revolutionized natural language processing with their transformative and emergent capabilities. However, recent evidence indicates that LLMs can produce harmful content that violates social norms, raising significant concerns regarding the safety and ethical ramifications of deploying these advanced models. Thus, it is both critical and imperative to perform a rigorous and comprehensive safety evaluation of LLMs before deployment. Despite this need, owing to the extensiveness of LLM generation space, it still lacks a unified and standardized risk ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14191","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-05-23T05:34:31Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"1ac4b6fdabc393ba0d3f997ad0267abb2330cb21725c5fbdda86ef00a524603a","abstract_canon_sha256":"5465c7140eee81b680e7c19a7a4a7b03edb452c902c46ae1fc2ad1c3d333813d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:08.244424Z","signature_b64":"343FFp6VjaQjsvri7udm8hYIqrs0GPKh78xWZ/bk1omo6KHboX5AbMJtVYbEFQnSo9+Dotc0n6JEvaKO8pfcDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22856bfbcb78c4d3a5cf7fd80416c553d6e91f89f439a56c1256acdaec92dd42","last_reissued_at":"2026-07-05T10:45:08.243925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:08.243925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"S-Eval: Towards Automated and Comprehensive Safety Evaluation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CR","authors_text":"Dongxia Wang, Hui Xue, Jialuo Chen, Jinfeng Li, Jingyi Wang, Kui Ren, Longtao Huang, Wenhai Wang, Xiaofeng Mao, Xiaohan Yuan, Xiaoxia Liu, Yuefeng Chen","submitted_at":"2024-05-23T05:34:31Z","abstract_excerpt":"Generative large language models (LLMs) have revolutionized natural language processing with their transformative and emergent capabilities. However, recent evidence indicates that LLMs can produce harmful content that violates social norms, raising significant concerns regarding the safety and ethical ramifications of deploying these advanced models. Thus, it is both critical and imperative to perform a rigorous and comprehensive safety evaluation of LLMs before deployment. Despite this need, owing to the extensiveness of LLM generation space, it still lacks a unified and standardized risk ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14191","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14191/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14191","created_at":"2026-07-05T10:45:08.243982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14191v4","created_at":"2026-07-05T10:45:08.243982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14191","created_at":"2026-07-05T10:45:08.243982+00:00"},{"alias_kind":"pith_short_12","alias_value":"EKCWX66LPDCN","created_at":"2026-07-05T10:45:08.243982+00:00"},{"alias_kind":"pith_short_16","alias_value":"EKCWX66LPDCNHJOP","created_at":"2026-07-05T10:45:08.243982+00:00"},{"alias_kind":"pith_short_8","alias_value":"EKCWX66L","created_at":"2026-07-05T10:45:08.243982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":4,"sample":[{"citing_arxiv_id":"2606.07535","citing_title":"Multilingual Refusal Alignment for Safer Large Language Models","ref_index":72,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27632","citing_title":"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2508.11222","citing_title":"ORFuzz: Fuzzing the \"Other Side\" of LLM Safety -- Testing Over-Refusal","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2404.16130","citing_title":"From Local to Global: A Graph RAG Approach to Query-Focused Summarization","ref_index":74,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP","json":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP.json","graph_json":"https://pith.science/api/pith-number/EKCWX66LPDCNHJOPP7MAIFWFKP/graph.json","events_json":"https://pith.science/api/pith-number/EKCWX66LPDCNHJOPP7MAIFWFKP/events.json","paper":"https://pith.science/paper/EKCWX66L"},"agent_actions":{"view_html":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP","download_json":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP.json","view_paper":"https://pith.science/paper/EKCWX66L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14191&json=true","fetch_graph":"https://pith.science/api/pith-number/EKCWX66LPDCNHJOPP7MAIFWFKP/graph.json","fetch_events":"https://pith.science/api/pith-number/EKCWX66LPDCNHJOPP7MAIFWFKP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP/action/storage_attestation","attest_author":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP/action/author_attestation","sign_citation":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP/action/citation_signature","submit_replication":"https://pith.science/pith/EKCWX66LPDCNHJOPP7MAIFWFKP/action/replication_record"}},"created_at":"2026-07-05T10:45:08.243982+00:00","updated_at":"2026-07-05T10:45:08.243982+00:00"}