{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:55FPSPI4GITVMYCD3FWFWOLZID","short_pith_number":"pith:55FPSPI4","schema_version":"1.0","canonical_sha256":"ef4af93d1c3227566043d96c5b397940df39eda5450bace9a0e6cf45b102eaf9","source":{"kind":"arxiv","id":"2403.12316","version":1},"attestation_state":"computed","paper":{"title":"OpenEval: Benchmarking Chinese LLMs across Capability, Alignment and Safety","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuang Liu, Deyi Xiong, Hongying Zan, Jiaxuan Li, Jinwang Song, Junhui Zhang, Ling Shi, Linhao Yu, Renren Jin, Sun Li, Tao Liu, Tingting Cui, Xinmeng Ji, Yufei Huang","submitted_at":"2024-03-18T23:21:37Z","abstract_excerpt":"The rapid development of Chinese large language models (LLMs) poses big challenges for efficient LLM evaluation. While current initiatives have introduced new benchmarks or evaluation platforms for assessing Chinese LLMs, many of these focus primarily on capabilities, usually overlooking potential alignment and safety issues. To address this gap, we introduce OpenEval, an evaluation testbed that benchmarks Chinese LLMs across capability, alignment and safety. For capability assessment, we include 12 benchmark datasets to evaluate Chinese LLMs from 4 sub-dimensions: NLP tasks, disciplinary know"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.12316","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-18T23:21:37Z","cross_cats_sorted":[],"title_canon_sha256":"64ce633a77bb75231a07ccc28c58990999a1d2a956b607889d7ccc04bfa29b1b","abstract_canon_sha256":"ec6de47f69097e2757756eea9e9907da6500aa601fd6401b78f652cc6743883a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:57:58.987705Z","signature_b64":"tdggTXlEJhTjQxwI6wWdDgQEiMtxjlDZW7O6kTqsWw2mKEhnuxggUIUEniHCTGvENHMyjZVNQRM+5Cfma0Q6Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef4af93d1c3227566043d96c5b397940df39eda5450bace9a0e6cf45b102eaf9","last_reissued_at":"2026-07-05T07:57:58.987150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:57:58.987150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenEval: Benchmarking Chinese LLMs across Capability, Alignment and Safety","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuang Liu, Deyi Xiong, Hongying Zan, Jiaxuan Li, Jinwang Song, Junhui Zhang, Ling Shi, Linhao Yu, Renren Jin, Sun Li, Tao Liu, Tingting Cui, Xinmeng Ji, Yufei Huang","submitted_at":"2024-03-18T23:21:37Z","abstract_excerpt":"The rapid development of Chinese large language models (LLMs) poses big challenges for efficient LLM evaluation. While current initiatives have introduced new benchmarks or evaluation platforms for assessing Chinese LLMs, many of these focus primarily on capabilities, usually overlooking potential alignment and safety issues. To address this gap, we introduce OpenEval, an evaluation testbed that benchmarks Chinese LLMs across capability, alignment and safety. For capability assessment, we include 12 benchmark datasets to evaluate Chinese LLMs from 4 sub-dimensions: NLP tasks, disciplinary know"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.12316","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.12316/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.12316","created_at":"2026-07-05T07:57:58.987212+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.12316v1","created_at":"2026-07-05T07:57:58.987212+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.12316","created_at":"2026-07-05T07:57:58.987212+00:00"},{"alias_kind":"pith_short_12","alias_value":"55FPSPI4GITV","created_at":"2026-07-05T07:57:58.987212+00:00"},{"alias_kind":"pith_short_16","alias_value":"55FPSPI4GITVMYCD","created_at":"2026-07-05T07:57:58.987212+00:00"},{"alias_kind":"pith_short_8","alias_value":"55FPSPI4","created_at":"2026-07-05T07:57:58.987212+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.14497","citing_title":"Star-Agents: Automatic Data Optimization with LLM Agents for Instruction Tuning","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID","json":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID.json","graph_json":"https://pith.science/api/pith-number/55FPSPI4GITVMYCD3FWFWOLZID/graph.json","events_json":"https://pith.science/api/pith-number/55FPSPI4GITVMYCD3FWFWOLZID/events.json","paper":"https://pith.science/paper/55FPSPI4"},"agent_actions":{"view_html":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID","download_json":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID.json","view_paper":"https://pith.science/paper/55FPSPI4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.12316&json=true","fetch_graph":"https://pith.science/api/pith-number/55FPSPI4GITVMYCD3FWFWOLZID/graph.json","fetch_events":"https://pith.science/api/pith-number/55FPSPI4GITVMYCD3FWFWOLZID/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID/action/timestamp_anchor","attest_storage":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID/action/storage_attestation","attest_author":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID/action/author_attestation","sign_citation":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID/action/citation_signature","submit_replication":"https://pith.science/pith/55FPSPI4GITVMYCD3FWFWOLZID/action/replication_record"}},"created_at":"2026-07-05T07:57:58.987212+00:00","updated_at":"2026-07-05T07:57:58.987212+00:00"}