{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XJ5LJAZE3UDF6HMGLEDQCHZCRP","short_pith_number":"pith:XJ5LJAZE","schema_version":"1.0","canonical_sha256":"ba7ab48324dd065f1d865907011f228bf21acab4181b36537e961394035c34ab","source":{"kind":"arxiv","id":"2505.14107","version":4},"attestation_state":"computed","paper":{"title":"DiagnosisArena: Benchmarking Diagnostic Reasoning for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiaji Liu, Linjie Mu, Pengfei Liu, Shaoting Zhang, Wei Nie, Xiaofan Zhang, Yakun Zhu, Yutong Huang, Zhongzhen Huang","submitted_at":"2025-05-20T09:14:53Z","abstract_excerpt":"The emergence of groundbreaking large language models capable of performing complex reasoning tasks holds significant promise for addressing various scientific challenges, including those arising in complex clinical scenarios. To enable their safe and effective deployment in real-world healthcare settings, it is urgently necessary to benchmark the diagnostic capabilities of current models systematically. Given the limitations of existing medical benchmarks in evaluating advanced diagnostic reasoning, we present DiagnosisArena, a comprehensive and challenging benchmark designed to rigorously as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14107","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-20T09:14:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b99ad760c0f955d71df028a691400db6816877b6d2b7b6682f807e5955798b94","abstract_canon_sha256":"677204a5dfc40893af5b10ddff72f6ab57d3043198c8caa6fa9654c09a7c763c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:45.773231Z","signature_b64":"mtPCsXP/GYn/NEJGYcHrH0iv5u5FQduslDm2NJXdWQuyNmsusz2XdcmWxnDKVy+nVpVL5apTGUFBoAp/R2oWAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba7ab48324dd065f1d865907011f228bf21acab4181b36537e961394035c34ab","last_reissued_at":"2026-07-05T11:11:45.772747Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:45.772747Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DiagnosisArena: Benchmarking Diagnostic Reasoning for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jiaji Liu, Linjie Mu, Pengfei Liu, Shaoting Zhang, Wei Nie, Xiaofan Zhang, Yakun Zhu, Yutong Huang, Zhongzhen Huang","submitted_at":"2025-05-20T09:14:53Z","abstract_excerpt":"The emergence of groundbreaking large language models capable of performing complex reasoning tasks holds significant promise for addressing various scientific challenges, including those arising in complex clinical scenarios. To enable their safe and effective deployment in real-world healthcare settings, it is urgently necessary to benchmark the diagnostic capabilities of current models systematically. Given the limitations of existing medical benchmarks in evaluating advanced diagnostic reasoning, we present DiagnosisArena, a comprehensive and challenging benchmark designed to rigorously as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14107","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14107","created_at":"2026-07-05T11:11:45.772802+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14107v4","created_at":"2026-07-05T11:11:45.772802+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14107","created_at":"2026-07-05T11:11:45.772802+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJ5LJAZE3UDF","created_at":"2026-07-05T11:11:45.772802+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJ5LJAZE3UDF6HMG","created_at":"2026-07-05T11:11:45.772802+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJ5LJAZE","created_at":"2026-07-05T11:11:45.772802+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00147","citing_title":"RareDxR1: Autonomous Medical Reasoning for Rare Disease Diagnosis Beyond Human Annotation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07853","citing_title":"Beyond English benchmarks: clinical llm evaluation in Brazilian Portuguese","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30637","citing_title":"EHRBench: An Automated and Reliable EHR-based Benchmark for Clinical Decision Making with LLMs","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22258","citing_title":"Beyond Classification Accuracy: Neural-MedBench and the Need for Deeper Reasoning Benchmarks","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12705","citing_title":"MedXIAOHE: A Comprehensive Recipe for Building Medical MLLMs","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09505","citing_title":"EpiGraph: Building Generalists for Evidence-Intensive Epilepsy Reasoning in the Wild","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09505","citing_title":"EpiGraph: Building Generalists for Evidence-Intensive Epilepsy Reasoning in the Wild","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06262","citing_title":"From Exposure to Internalization: Dual-Stream Calibration for In-context Clinical Reasoning","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP","json":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP.json","graph_json":"https://pith.science/api/pith-number/XJ5LJAZE3UDF6HMGLEDQCHZCRP/graph.json","events_json":"https://pith.science/api/pith-number/XJ5LJAZE3UDF6HMGLEDQCHZCRP/events.json","paper":"https://pith.science/paper/XJ5LJAZE"},"agent_actions":{"view_html":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP","download_json":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP.json","view_paper":"https://pith.science/paper/XJ5LJAZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14107&json=true","fetch_graph":"https://pith.science/api/pith-number/XJ5LJAZE3UDF6HMGLEDQCHZCRP/graph.json","fetch_events":"https://pith.science/api/pith-number/XJ5LJAZE3UDF6HMGLEDQCHZCRP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP/action/storage_attestation","attest_author":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP/action/author_attestation","sign_citation":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP/action/citation_signature","submit_replication":"https://pith.science/pith/XJ5LJAZE3UDF6HMGLEDQCHZCRP/action/replication_record"}},"created_at":"2026-07-05T11:11:45.772802+00:00","updated_at":"2026-07-05T11:11:45.772802+00:00"}