{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CZ4X2EHNWEXC5I6VUO6CLPYNLO","short_pith_number":"pith:CZ4X2EHN","schema_version":"1.0","canonical_sha256":"16797d10edb12e2ea3d5a3bc25bf0d5b8b6159cfe2515b85e76d1ebc6afba66c","source":{"kind":"arxiv","id":"2506.06636","version":1},"attestation_state":"computed","paper":{"title":"SafeLawBench: Towards Safe Alignment of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuxue Cao, Han Zhu, Jiaming Ji, Juntao Dai, QiChao Sun, Sirui Han, Yaodong Yang, Yike Guo, Yinyu Wu, Zhenghao Zhu","submitted_at":"2025-06-07T03:09:59Z","abstract_excerpt":"With the growing prevalence of large language models (LLMs), the safety of LLMs has raised significant concerns. However, there is still a lack of definitive standards for evaluating their safety due to the subjective nature of current safety benchmarks. To address this gap, we conducted the first exploration of LLMs' safety evaluation from a legal perspective by proposing the SafeLawBench benchmark. SafeLawBench categorizes safety risks into three levels based on legal standards, providing a systematic and comprehensive framework for evaluation. It comprises 24,860 multi-choice questions and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.06636","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-07T03:09:59Z","cross_cats_sorted":[],"title_canon_sha256":"a4fc605b4cb07b3d165ccda1ccc1c3d2e1d61621e15048143a1e94a335b65c4e","abstract_canon_sha256":"26554ccbeb87691e8a95022e1ad9edefdb3ef459bb8aa2ab97664bcfe2977647"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:53.255715Z","signature_b64":"9Spe9wgJBYfAzLJH3t9/ryFKbbUDYdFOJn7bPf/YUxldw7zRTs5OjyZA1D/zaywBKJUrOsER1iCFQNHLL7yZBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16797d10edb12e2ea3d5a3bc25bf0d5b8b6159cfe2515b85e76d1ebc6afba66c","last_reissued_at":"2026-07-05T11:17:53.255189Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:53.255189Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SafeLawBench: Towards Safe Alignment of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuxue Cao, Han Zhu, Jiaming Ji, Juntao Dai, QiChao Sun, Sirui Han, Yaodong Yang, Yike Guo, Yinyu Wu, Zhenghao Zhu","submitted_at":"2025-06-07T03:09:59Z","abstract_excerpt":"With the growing prevalence of large language models (LLMs), the safety of LLMs has raised significant concerns. However, there is still a lack of definitive standards for evaluating their safety due to the subjective nature of current safety benchmarks. To address this gap, we conducted the first exploration of LLMs' safety evaluation from a legal perspective by proposing the SafeLawBench benchmark. SafeLawBench categorizes safety risks into three levels based on legal standards, providing a systematic and comprehensive framework for evaluation. It comprises 24,860 multi-choice questions and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.06636","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.06636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.06636","created_at":"2026-07-05T11:17:53.255259+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.06636v1","created_at":"2026-07-05T11:17:53.255259+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.06636","created_at":"2026-07-05T11:17:53.255259+00:00"},{"alias_kind":"pith_short_12","alias_value":"CZ4X2EHNWEXC","created_at":"2026-07-05T11:17:53.255259+00:00"},{"alias_kind":"pith_short_16","alias_value":"CZ4X2EHNWEXC5I6V","created_at":"2026-07-05T11:17:53.255259+00:00"},{"alias_kind":"pith_short_8","alias_value":"CZ4X2EHN","created_at":"2026-07-05T11:17:53.255259+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00245","citing_title":"ARMOR 2025: A Military-Aligned Benchmark for Evaluating Large Language Model Safety Beyond Civilian Contexts","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO","json":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO.json","graph_json":"https://pith.science/api/pith-number/CZ4X2EHNWEXC5I6VUO6CLPYNLO/graph.json","events_json":"https://pith.science/api/pith-number/CZ4X2EHNWEXC5I6VUO6CLPYNLO/events.json","paper":"https://pith.science/paper/CZ4X2EHN"},"agent_actions":{"view_html":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO","download_json":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO.json","view_paper":"https://pith.science/paper/CZ4X2EHN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.06636&json=true","fetch_graph":"https://pith.science/api/pith-number/CZ4X2EHNWEXC5I6VUO6CLPYNLO/graph.json","fetch_events":"https://pith.science/api/pith-number/CZ4X2EHNWEXC5I6VUO6CLPYNLO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO/action/storage_attestation","attest_author":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO/action/author_attestation","sign_citation":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO/action/citation_signature","submit_replication":"https://pith.science/pith/CZ4X2EHNWEXC5I6VUO6CLPYNLO/action/replication_record"}},"created_at":"2026-07-05T11:17:53.255259+00:00","updated_at":"2026-07-05T11:17:53.255259+00:00"}