{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CYEOKRTXPFOZBMYGAIVIMOLG5F","short_pith_number":"pith:CYEOKRTX","schema_version":"1.0","canonical_sha256":"1608e54677795d90b306022a863966e96a6dfd1532950f936a3c661f6d6fdaa2","source":{"kind":"arxiv","id":"2412.15265","version":2},"attestation_state":"computed","paper":{"title":"Chinese SafetyQA: A Safety Short-form Factuality Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baihui Zheng, Boren Zheng, Bo Zheng, Huiyun Jing, Jiaheng Liu, Jincheng Wei, Kaifu Zhang, Kerui Cao, Wenbo Su, Xiangyong Zhu, Yancheng He, Yingshui Tan","submitted_at":"2024-12-17T03:03:44Z","abstract_excerpt":"With the rapid advancement of Large Language Models (LLMs), significant safety concerns have emerged. Fundamentally, the safety of large language models is closely linked to the accuracy, comprehensiveness, and clarity of their understanding of safety knowledge, particularly in domains such as law, policy and ethics. This factuality ability is crucial in determining whether these models can be deployed and applied safely and compliantly within specific regions. To address these challenges and better evaluate the factuality ability of LLMs to answer short questions, we introduce the Chinese Saf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15265","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-17T03:03:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"de011a7bdf66d3a22031306fb816ec8a2c6763bcd9a28eb0c8cbd804d95050dc","abstract_canon_sha256":"1045732e739eb533fab9d2c5ac5e64e28f35bb0c974e98183d7a8f9c9bd68570"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:53:09.621844Z","signature_b64":"deud5IyueTMg5nCBLn8hvo8PMUdz2EiiTi4vsUuUaIlEU0JGcWvYhjfK/BSTUaumGMf7q7P5S0lNanoJ2mN4Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1608e54677795d90b306022a863966e96a6dfd1532950f936a3c661f6d6fdaa2","last_reissued_at":"2026-07-05T09:53:09.621386Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:53:09.621386Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chinese SafetyQA: A Safety Short-form Factuality Benchmark for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baihui Zheng, Boren Zheng, Bo Zheng, Huiyun Jing, Jiaheng Liu, Jincheng Wei, Kaifu Zhang, Kerui Cao, Wenbo Su, Xiangyong Zhu, Yancheng He, Yingshui Tan","submitted_at":"2024-12-17T03:03:44Z","abstract_excerpt":"With the rapid advancement of Large Language Models (LLMs), significant safety concerns have emerged. Fundamentally, the safety of large language models is closely linked to the accuracy, comprehensiveness, and clarity of their understanding of safety knowledge, particularly in domains such as law, policy and ethics. This factuality ability is crucial in determining whether these models can be deployed and applied safely and compliantly within specific regions. To address these challenges and better evaluate the factuality ability of LLMs to answer short questions, we introduce the Chinese Saf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15265","created_at":"2026-07-05T09:53:09.621442+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15265v2","created_at":"2026-07-05T09:53:09.621442+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15265","created_at":"2026-07-05T09:53:09.621442+00:00"},{"alias_kind":"pith_short_12","alias_value":"CYEOKRTXPFOZ","created_at":"2026-07-05T09:53:09.621442+00:00"},{"alias_kind":"pith_short_16","alias_value":"CYEOKRTXPFOZBMYG","created_at":"2026-07-05T09:53:09.621442+00:00"},{"alias_kind":"pith_short_8","alias_value":"CYEOKRTX","created_at":"2026-07-05T09:53:09.621442+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.04975","citing_title":"Evaluating Chinese Large Language Models: The Influence of Persona Assignment on Stereotypes and Safeguards","ref_index":53,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F","json":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F.json","graph_json":"https://pith.science/api/pith-number/CYEOKRTXPFOZBMYGAIVIMOLG5F/graph.json","events_json":"https://pith.science/api/pith-number/CYEOKRTXPFOZBMYGAIVIMOLG5F/events.json","paper":"https://pith.science/paper/CYEOKRTX"},"agent_actions":{"view_html":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F","download_json":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F.json","view_paper":"https://pith.science/paper/CYEOKRTX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15265&json=true","fetch_graph":"https://pith.science/api/pith-number/CYEOKRTXPFOZBMYGAIVIMOLG5F/graph.json","fetch_events":"https://pith.science/api/pith-number/CYEOKRTXPFOZBMYGAIVIMOLG5F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F/action/storage_attestation","attest_author":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F/action/author_attestation","sign_citation":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F/action/citation_signature","submit_replication":"https://pith.science/pith/CYEOKRTXPFOZBMYGAIVIMOLG5F/action/replication_record"}},"created_at":"2026-07-05T09:53:09.621442+00:00","updated_at":"2026-07-05T09:53:09.621442+00:00"}