{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WLTHTYVQ772AXMWVMW5QNIIAID","short_pith_number":"pith:WLTHTYVQ","schema_version":"1.0","canonical_sha256":"b2e679e2b0fff40bb2d565bb06a10040f263c8321c08eeb4d8c3f2b5dc161ccf","source":{"kind":"arxiv","id":"2505.17047","version":1},"attestation_state":"computed","paper":{"title":"Assessing the Quality of AI-Generated Clinical Notes: A Validated Evaluation of a Large Language Model Scribe","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Astrit Manikantan, Erin Palm, Herprit Mahal, Mark E. Pepin, Srikanth Subramanya Belwadi","submitted_at":"2025-05-15T16:14:53Z","abstract_excerpt":"In medical practices across the United States, physicians have begun implementing generative artificial intelligence (AI) tools to perform the function of scribes in order to reduce the burden of documenting clinical encounters. Despite their widespread use, no established methods exist to gauge the quality of AI scribes. To address this gap, we developed a blinded study comparing the relative performance of large language model (LLM) generated clinical notes with those from field experts based on audio-recorded clinical encounters. Quantitative metrics from the Physician Documentation Quality"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17047","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-15T16:14:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8e577d4a2bb09b4b2281f71a6f0b30aac60ac5c5276971fa8927ee022dfce649","abstract_canon_sha256":"53fbb5eb833d25b5e214be846924cf7184fe76832e000e458f1a98c62d087056"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:46.165206Z","signature_b64":"DapdP8wmeivKBLPsGX8l+e6q0yC0nSFnMkvPhUxOrRniULurFaJOzjyExjbbiAxqDhpmNGt0n/7++xaGY0CuDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2e679e2b0fff40bb2d565bb06a10040f263c8321c08eeb4d8c3f2b5dc161ccf","last_reissued_at":"2026-07-05T11:07:46.164666Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:46.164666Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assessing the Quality of AI-Generated Clinical Notes: A Validated Evaluation of a Large Language Model Scribe","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Astrit Manikantan, Erin Palm, Herprit Mahal, Mark E. Pepin, Srikanth Subramanya Belwadi","submitted_at":"2025-05-15T16:14:53Z","abstract_excerpt":"In medical practices across the United States, physicians have begun implementing generative artificial intelligence (AI) tools to perform the function of scribes in order to reduce the burden of documenting clinical encounters. Despite their widespread use, no established methods exist to gauge the quality of AI scribes. To address this gap, we developed a blinded study comparing the relative performance of large language model (LLM) generated clinical notes with those from field experts based on audio-recorded clinical encounters. Quantitative metrics from the Physician Documentation Quality"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17047","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17047","created_at":"2026-07-05T11:07:46.164743+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17047v1","created_at":"2026-07-05T11:07:46.164743+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17047","created_at":"2026-07-05T11:07:46.164743+00:00"},{"alias_kind":"pith_short_12","alias_value":"WLTHTYVQ772A","created_at":"2026-07-05T11:07:46.164743+00:00"},{"alias_kind":"pith_short_16","alias_value":"WLTHTYVQ772AXMWV","created_at":"2026-07-05T11:07:46.164743+00:00"},{"alias_kind":"pith_short_8","alias_value":"WLTHTYVQ","created_at":"2026-07-05T11:07:46.164743+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.14079","citing_title":"DENSE: Longitudinal Progress Note Generation with Temporal Modeling of Heterogeneous Clinical Notes Across Hospital Visits","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID","json":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID.json","graph_json":"https://pith.science/api/pith-number/WLTHTYVQ772AXMWVMW5QNIIAID/graph.json","events_json":"https://pith.science/api/pith-number/WLTHTYVQ772AXMWVMW5QNIIAID/events.json","paper":"https://pith.science/paper/WLTHTYVQ"},"agent_actions":{"view_html":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID","download_json":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID.json","view_paper":"https://pith.science/paper/WLTHTYVQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17047&json=true","fetch_graph":"https://pith.science/api/pith-number/WLTHTYVQ772AXMWVMW5QNIIAID/graph.json","fetch_events":"https://pith.science/api/pith-number/WLTHTYVQ772AXMWVMW5QNIIAID/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID/action/storage_attestation","attest_author":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID/action/author_attestation","sign_citation":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID/action/citation_signature","submit_replication":"https://pith.science/pith/WLTHTYVQ772AXMWVMW5QNIIAID/action/replication_record"}},"created_at":"2026-07-05T11:07:46.164743+00:00","updated_at":"2026-07-05T11:07:46.164743+00:00"}