{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LBSTRZCLCFJRCYJ3PNDM3QWGQQ","short_pith_number":"pith:LBSTRZCL","schema_version":"1.0","canonical_sha256":"586538e44b115311613b7b46cdc2c6843fd4d5bfe743e587a360256619aa7431","source":{"kind":"arxiv","id":"2503.04691","version":2},"attestation_state":"computed","paper":{"title":"Quantifying the Reasoning Abilities of LLMs on Real-world Clinical Cases","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chaoyi Wu, Chuanjin Peng, Hongfei Gu, Pengcheng Qiu, Shuyu Liu, Weidi Xie, Weike Zhao, Yanfeng Wang, Ya Zhang, Zhuoxia Chen","submitted_at":"2025-03-06T18:35:39Z","abstract_excerpt":"Recent advancements in reasoning-enhanced large language models (LLMs), such as DeepSeek-R1 and OpenAI-o3, have demonstrated significant progress. However, their application in professional medical contexts remains underexplored, particularly in evaluating the quality of their reasoning processes alongside final outputs. Here, we introduce MedR-Bench, a benchmarking dataset of 1,453 structured patient cases, annotated with reasoning references derived from clinical case reports. Spanning 13 body systems and 10 specialties, it includes both common and rare diseases. To comprehensively evaluate "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.04691","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-06T18:35:39Z","cross_cats_sorted":[],"title_canon_sha256":"5b685375c830393002b84d09fbf7e192f013cdc3c8fd439df12f71661a7fe945","abstract_canon_sha256":"48ee0e488769caa498c160d206cdfe41f10201f41411a76df054ff52d2ec290a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:05.833292Z","signature_b64":"1c1/RmCuol+fWNmvRzsHKsA0/BshaAoeoqNminVUEyjSLOwHsBvcaVdKesZMGlr3kup4L6iwanhP48Wyuzt2Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"586538e44b115311613b7b46cdc2c6843fd4d5bfe743e587a360256619aa7431","last_reissued_at":"2026-07-05T10:28:05.832630Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:05.832630Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying the Reasoning Abilities of LLMs on Real-world Clinical Cases","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chaoyi Wu, Chuanjin Peng, Hongfei Gu, Pengcheng Qiu, Shuyu Liu, Weidi Xie, Weike Zhao, Yanfeng Wang, Ya Zhang, Zhuoxia Chen","submitted_at":"2025-03-06T18:35:39Z","abstract_excerpt":"Recent advancements in reasoning-enhanced large language models (LLMs), such as DeepSeek-R1 and OpenAI-o3, have demonstrated significant progress. However, their application in professional medical contexts remains underexplored, particularly in evaluating the quality of their reasoning processes alongside final outputs. Here, we introduce MedR-Bench, a benchmarking dataset of 1,453 structured patient cases, annotated with reasoning references derived from clinical case reports. Spanning 13 body systems and 10 specialties, it includes both common and rare diseases. To comprehensively evaluate "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.04691","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.04691/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.04691","created_at":"2026-07-05T10:28:05.832702+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.04691v2","created_at":"2026-07-05T10:28:05.832702+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.04691","created_at":"2026-07-05T10:28:05.832702+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBSTRZCLCFJR","created_at":"2026-07-05T10:28:05.832702+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBSTRZCLCFJRCYJ3","created_at":"2026-07-05T10:28:05.832702+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBSTRZCL","created_at":"2026-07-05T10:28:05.832702+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07761","citing_title":"Aligning Clinical Needs and AI Capabilities: A Survey on LLMs for Medical Reasoning","ref_index":153,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03157","citing_title":"ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01094","citing_title":"CAREAgent: Clinical Agent with Structured Reasoning and Tool-Integrated for Order Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29746","citing_title":"DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30637","citing_title":"EHRBench: An Automated and Reliable EHR-based Benchmark for Clinical Decision Making with LLMs","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2505.14558","citing_title":"R2MED: A Benchmark for Reasoning-Driven Medical Retrieval","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09584","citing_title":"CLR-voyance: Reinforcing Open-Ended Reasoning for Inpatient Clinical Decision Support with Outcome-Aware Rubrics","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ","json":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ.json","graph_json":"https://pith.science/api/pith-number/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/graph.json","events_json":"https://pith.science/api/pith-number/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/events.json","paper":"https://pith.science/paper/LBSTRZCL"},"agent_actions":{"view_html":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ","download_json":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ.json","view_paper":"https://pith.science/paper/LBSTRZCL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.04691&json=true","fetch_graph":"https://pith.science/api/pith-number/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/graph.json","fetch_events":"https://pith.science/api/pith-number/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/action/storage_attestation","attest_author":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/action/author_attestation","sign_citation":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/action/citation_signature","submit_replication":"https://pith.science/pith/LBSTRZCLCFJRCYJ3PNDM3QWGQQ/action/replication_record"}},"created_at":"2026-07-05T10:28:05.832702+00:00","updated_at":"2026-07-05T10:28:05.832702+00:00"}