{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3SE2YS3CLPLUBJAGENN2XJ54SZ","short_pith_number":"pith:3SE2YS3C","schema_version":"1.0","canonical_sha256":"dc89ac4b625bd740a406235baba7bc96603fd5e0144800b9f4706f4055fe38d5","source":{"kind":"arxiv","id":"2402.16040","version":5},"attestation_state":"computed","paper":{"title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dongchul Cha, Edward Choi, Hangyul Yoon, Heeyoung Kwak, Jeewon Yang, Jiyoun Kim, Kwanghyun Kim, Seunghyun Won, Sunjun Kweon","submitted_at":"2024-02-25T09:41:50Z","abstract_excerpt":"Discharge summaries in Electronic Health Records (EHRs) are crucial for clinical decision-making, but their length and complexity make information extraction challenging, especially when dealing with accumulated summaries across multiple patient admissions. Large Language Models (LLMs) show promise in addressing this challenge by efficiently analyzing vast and complex data. Existing benchmarks, however, fall short in properly evaluating LLMs' capabilities in this context, as they typically focus on single-note information or limited topics, failing to reflect the real-world inquiries required "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.16040","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-25T09:41:50Z","cross_cats_sorted":[],"title_canon_sha256":"9603a65d1b59d64aa8f8ffb9071e082aa6a8666b48d95ff8723b5c62a007efa8","abstract_canon_sha256":"d5d8cbf0ef64074155a6d7aff4bc61c1310f89ec69170ed9ef1377d145145c45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:29.047663Z","signature_b64":"VdOJbROvQoOBqUO89iyrmHpDA/rltd3MygYSYeldTl1JyfFmfX9ctrvXGd2erJKzLiHxNs6kHPls7osvzc70Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc89ac4b625bd740a406235baba7bc96603fd5e0144800b9f4706f4055fe38d5","last_reissued_at":"2026-07-05T09:33:29.047115Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:29.047115Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dongchul Cha, Edward Choi, Hangyul Yoon, Heeyoung Kwak, Jeewon Yang, Jiyoun Kim, Kwanghyun Kim, Seunghyun Won, Sunjun Kweon","submitted_at":"2024-02-25T09:41:50Z","abstract_excerpt":"Discharge summaries in Electronic Health Records (EHRs) are crucial for clinical decision-making, but their length and complexity make information extraction challenging, especially when dealing with accumulated summaries across multiple patient admissions. Large Language Models (LLMs) show promise in addressing this challenge by efficiently analyzing vast and complex data. Existing benchmarks, however, fall short in properly evaluating LLMs' capabilities in this context, as they typically focus on single-note information or limited topics, failing to reflect the real-world inquiries required "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16040","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.16040/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.16040","created_at":"2026-07-05T09:33:29.047169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.16040v5","created_at":"2026-07-05T09:33:29.047169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16040","created_at":"2026-07-05T09:33:29.047169+00:00"},{"alias_kind":"pith_short_12","alias_value":"3SE2YS3CLPLU","created_at":"2026-07-05T09:33:29.047169+00:00"},{"alias_kind":"pith_short_16","alias_value":"3SE2YS3CLPLUBJAG","created_at":"2026-07-05T09:33:29.047169+00:00"},{"alias_kind":"pith_short_8","alias_value":"3SE2YS3C","created_at":"2026-07-05T09:33:29.047169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18471","citing_title":"Possible or Definite? A Benchmark for Evaluating Diagnostic Uncertainty Preservation in Clinical Text","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":170,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ","json":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ.json","graph_json":"https://pith.science/api/pith-number/3SE2YS3CLPLUBJAGENN2XJ54SZ/graph.json","events_json":"https://pith.science/api/pith-number/3SE2YS3CLPLUBJAGENN2XJ54SZ/events.json","paper":"https://pith.science/paper/3SE2YS3C"},"agent_actions":{"view_html":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ","download_json":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ.json","view_paper":"https://pith.science/paper/3SE2YS3C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.16040&json=true","fetch_graph":"https://pith.science/api/pith-number/3SE2YS3CLPLUBJAGENN2XJ54SZ/graph.json","fetch_events":"https://pith.science/api/pith-number/3SE2YS3CLPLUBJAGENN2XJ54SZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ/action/storage_attestation","attest_author":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ/action/author_attestation","sign_citation":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ/action/citation_signature","submit_replication":"https://pith.science/pith/3SE2YS3CLPLUBJAGENN2XJ54SZ/action/replication_record"}},"created_at":"2026-07-05T09:33:29.047169+00:00","updated_at":"2026-07-05T09:33:29.047169+00:00"}