{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GMXIGQ7234KIE26LQY7EEDPCR2","short_pith_number":"pith:GMXIGQ72","schema_version":"1.0","canonical_sha256":"332e8343fadf14826bcb863e420de28e9855b5e4d27422df253c99b00cb5ec86","source":{"kind":"arxiv","id":"2411.07140","version":2},"attestation_state":"computed","paper":{"title":"Chinese SimpleQA: A Chinese Factuality Evaluation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boren Zheng, Bo Zheng, Chengwei Hu, Dekai Sun, Hangyu Guo, Hui Huang, Jiaheng Liu, Shilong Li, Shirong Lin, Weixun Wang, Wenbo Su, Xiaoyong Zhu, Xingyuan Bu, Xuepeng Liu, Yancheng He, Yingshui Tan, Zhicheng Zheng, Zhuoran Lin","submitted_at":"2024-11-11T17:10:56Z","abstract_excerpt":"New LLM evaluation benchmarks are important to align with the rapid development of Large Language Models (LLMs). In this work, we present Chinese SimpleQA, the first comprehensive Chinese benchmark to evaluate the factuality ability of language models to answer short questions, and Chinese SimpleQA mainly has five properties (i.e., Chinese, Diverse, High-quality, Static, Easy-to-evaluate). Specifically, first, we focus on the Chinese language over 6 major topics with 99 diverse subtopics. Second, we conduct a comprehensive quality control process to achieve high-quality questions and answers, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.07140","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-11T17:10:56Z","cross_cats_sorted":[],"title_canon_sha256":"a9f0f7460b974f1a47cfae96b30b95b8459aa13c552d2a8dcf919079fe506d03","abstract_canon_sha256":"b55fe4bb1c3e3ed58274ffd1f7c0b69fd53946303ccd30474f6d97a7a7ea6eba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:58.696056Z","signature_b64":"Rd8woI63ZxdHgmK6kSc1mCJbHThOe3s4qbyYfYUcSuzHnRI5FzvZyHzOM7TcsYSr6M9kxjip6+kxtPSatyI6BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"332e8343fadf14826bcb863e420de28e9855b5e4d27422df253c99b00cb5ec86","last_reissued_at":"2026-07-05T09:34:58.695547Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:58.695547Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chinese SimpleQA: A Chinese Factuality Evaluation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Boren Zheng, Bo Zheng, Chengwei Hu, Dekai Sun, Hangyu Guo, Hui Huang, Jiaheng Liu, Shilong Li, Shirong Lin, Weixun Wang, Wenbo Su, Xiaoyong Zhu, Xingyuan Bu, Xuepeng Liu, Yancheng He, Yingshui Tan, Zhicheng Zheng, Zhuoran Lin","submitted_at":"2024-11-11T17:10:56Z","abstract_excerpt":"New LLM evaluation benchmarks are important to align with the rapid development of Large Language Models (LLMs). In this work, we present Chinese SimpleQA, the first comprehensive Chinese benchmark to evaluate the factuality ability of language models to answer short questions, and Chinese SimpleQA mainly has five properties (i.e., Chinese, Diverse, High-quality, Static, Easy-to-evaluate). Specifically, first, we focus on the Chinese language over 6 major topics with 99 diverse subtopics. Second, we conduct a comprehensive quality control process to achieve high-quality questions and answers, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.07140","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.07140/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.07140","created_at":"2026-07-05T09:34:58.695606+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.07140v2","created_at":"2026-07-05T09:34:58.695606+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.07140","created_at":"2026-07-05T09:34:58.695606+00:00"},{"alias_kind":"pith_short_12","alias_value":"GMXIGQ7234KI","created_at":"2026-07-05T09:34:58.695606+00:00"},{"alias_kind":"pith_short_16","alias_value":"GMXIGQ7234KIE26L","created_at":"2026-07-05T09:34:58.695606+00:00"},{"alias_kind":"pith_short_8","alias_value":"GMXIGQ72","created_at":"2026-07-05T09:34:58.695606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10722","citing_title":"Continual LLM Upcycling: A Predictor-Gated Bank-Wise Sparsity Training Recipe for Dense-to-Sparse LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2506.19807","citing_title":"KnowRL: Exploring Knowledgeable Reinforcement Learning for Factuality","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05497","citing_title":"Patterns behind Chaos: Forecasting Data Movement for Efficient Large-Scale MoE LLM Inference","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02805","citing_title":"MemSearcher: Training LLMs to Reason, Search and Manage Memory via End-to-End Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2502.03387","citing_title":"LIMO: Less is More for Reasoning","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2507.20534","citing_title":"Kimi K2: Open Agentic Intelligence","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2","json":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2.json","graph_json":"https://pith.science/api/pith-number/GMXIGQ7234KIE26LQY7EEDPCR2/graph.json","events_json":"https://pith.science/api/pith-number/GMXIGQ7234KIE26LQY7EEDPCR2/events.json","paper":"https://pith.science/paper/GMXIGQ72"},"agent_actions":{"view_html":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2","download_json":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2.json","view_paper":"https://pith.science/paper/GMXIGQ72","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.07140&json=true","fetch_graph":"https://pith.science/api/pith-number/GMXIGQ7234KIE26LQY7EEDPCR2/graph.json","fetch_events":"https://pith.science/api/pith-number/GMXIGQ7234KIE26LQY7EEDPCR2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2/action/storage_attestation","attest_author":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2/action/author_attestation","sign_citation":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2/action/citation_signature","submit_replication":"https://pith.science/pith/GMXIGQ7234KIE26LQY7EEDPCR2/action/replication_record"}},"created_at":"2026-07-05T09:34:58.695606+00:00","updated_at":"2026-07-05T09:34:58.695606+00:00"}