{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AMHKRN5UDRXFCLB62PRWWKQO3J","short_pith_number":"pith:AMHKRN5U","schema_version":"1.0","canonical_sha256":"030ea8b7b41c6e512c3ed3e36b2a0eda7db2c21c32187f5568715deb834e5bea","source":{"kind":"arxiv","id":"2507.23486","version":3},"attestation_state":"computed","paper":{"title":"A Novel Evaluation Benchmark for Medical LLMs: Illuminating Safety and Effectiveness in Clinical Domains","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chunbo Wang, Dongqiang Ye, Haitao Shen, Hongyang Ma, Hua Gao, Huaxia Yang, Huimin Zhang, Jianxiong Wu, Kehang Mao, Lijuan Niu, Lina Zhou, Lingmin Meng, Lingyun Ma, Long Chen, Mengran Lang, Mingyu Chen, Naixin Liang, Peng Yu, Qianling Ye, Qiguang Zhao, Qiuhong Gong, Rui Lin, Shirui Wang, Si-Xuan Liu, Tiantian Gu, Wenjie Shen, Wubin Sun, Yajie Ji, Yinan Jiang, Yongxin Wang, Youtao Yu, Yue Liu, Yunhui Tan, Yunlu Gao, Zeliang Lian, Zhicheng Huang, Zhihao Wang, Zhihui Tang","submitted_at":"2025-07-31T12:10:00Z","abstract_excerpt":"Large language models (LLMs) hold promise in clinical decision support but face major challenges in safety evaluation and effectiveness validation. We developed the Clinical Safety-Effectiveness Dual-Track Benchmark (CSEDB), a multidimensional framework built on clinical expert consensus, encompassing 30 criteria covering critical areas like critical illness recognition, guideline adherence, and medication safety, with weighted consequence measures. Thirty-two specialist physicians developed and reviewed 2,069 open-ended Q&A items aligned with these criteria, spanning 26 clinical departments t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.23486","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-07-31T12:10:00Z","cross_cats_sorted":[],"title_canon_sha256":"2c66fc8a0b74fff73a8c298bfd088891afe728c354701b9d4296a711086e35ca","abstract_canon_sha256":"7bf7a4b1340e122314cee77a68ce5dd56c16ba395f543f089a5a68b9ad1762ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:06.421018Z","signature_b64":"B1wIlsOMJH6Tq+kEEP00iNs2B+05ff5l2VQOl36dbsrH2aIh2+Q0rxbaINSFJH+X8HcpzogZzc0pEy/ZdomDBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"030ea8b7b41c6e512c3ed3e36b2a0eda7db2c21c32187f5568715deb834e5bea","last_reissued_at":"2026-07-05T11:53:06.420536Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:06.420536Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Novel Evaluation Benchmark for Medical LLMs: Illuminating Safety and Effectiveness in Clinical Domains","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chunbo Wang, Dongqiang Ye, Haitao Shen, Hongyang Ma, Hua Gao, Huaxia Yang, Huimin Zhang, Jianxiong Wu, Kehang Mao, Lijuan Niu, Lina Zhou, Lingmin Meng, Lingyun Ma, Long Chen, Mengran Lang, Mingyu Chen, Naixin Liang, Peng Yu, Qianling Ye, Qiguang Zhao, Qiuhong Gong, Rui Lin, Shirui Wang, Si-Xuan Liu, Tiantian Gu, Wenjie Shen, Wubin Sun, Yajie Ji, Yinan Jiang, Yongxin Wang, Youtao Yu, Yue Liu, Yunhui Tan, Yunlu Gao, Zeliang Lian, Zhicheng Huang, Zhihao Wang, Zhihui Tang","submitted_at":"2025-07-31T12:10:00Z","abstract_excerpt":"Large language models (LLMs) hold promise in clinical decision support but face major challenges in safety evaluation and effectiveness validation. We developed the Clinical Safety-Effectiveness Dual-Track Benchmark (CSEDB), a multidimensional framework built on clinical expert consensus, encompassing 30 criteria covering critical areas like critical illness recognition, guideline adherence, and medication safety, with weighted consequence measures. Thirty-two specialist physicians developed and reviewed 2,069 open-ended Q&A items aligned with these criteria, spanning 26 clinical departments t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.23486","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.23486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.23486","created_at":"2026-07-05T11:53:06.420593+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.23486v3","created_at":"2026-07-05T11:53:06.420593+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.23486","created_at":"2026-07-05T11:53:06.420593+00:00"},{"alias_kind":"pith_short_12","alias_value":"AMHKRN5UDRXF","created_at":"2026-07-05T11:53:06.420593+00:00"},{"alias_kind":"pith_short_16","alias_value":"AMHKRN5UDRXFCLB6","created_at":"2026-07-05T11:53:06.420593+00:00"},{"alias_kind":"pith_short_8","alias_value":"AMHKRN5U","created_at":"2026-07-05T11:53:06.420593+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12702","citing_title":"Deployment-Centered Evaluation: Predicting Query-Level Rejection Risk in a Clinical LLM System","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07919","citing_title":"MedVIGIL: Evaluating Trustworthy Medical VLMs Under Broken Visual Evidence","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07919","citing_title":"MedVIGIL: Evaluating Trustworthy Medical VLMs Under Broken Visual Evidence","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07709","citing_title":"IatroBench: Pre-Registered Evidence of Iatrogenic Harm from AI Safety Measures","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J","json":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J.json","graph_json":"https://pith.science/api/pith-number/AMHKRN5UDRXFCLB62PRWWKQO3J/graph.json","events_json":"https://pith.science/api/pith-number/AMHKRN5UDRXFCLB62PRWWKQO3J/events.json","paper":"https://pith.science/paper/AMHKRN5U"},"agent_actions":{"view_html":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J","download_json":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J.json","view_paper":"https://pith.science/paper/AMHKRN5U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.23486&json=true","fetch_graph":"https://pith.science/api/pith-number/AMHKRN5UDRXFCLB62PRWWKQO3J/graph.json","fetch_events":"https://pith.science/api/pith-number/AMHKRN5UDRXFCLB62PRWWKQO3J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J/action/storage_attestation","attest_author":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J/action/author_attestation","sign_citation":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J/action/citation_signature","submit_replication":"https://pith.science/pith/AMHKRN5UDRXFCLB62PRWWKQO3J/action/replication_record"}},"created_at":"2026-07-05T11:53:06.420593+00:00","updated_at":"2026-07-05T11:53:06.420593+00:00"}