{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DQRXDCVPCL7IQPZHK7J2CENR2R","short_pith_number":"pith:DQRXDCVP","schema_version":"1.0","canonical_sha256":"1c23718aaf12fe883f2757d3a111b1d459bf46b0529eace7eb105d9873fac689","source":{"kind":"arxiv","id":"2311.13273","version":2},"attestation_state":"computed","paper":{"title":"Comparative Experimentation of Accuracy Metrics in Automated Medical Reporting: The Case of Otitis Consultations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Renske Eline Bootsma, Sandra van Dulmen, Sjaak Brinkkemper, Tom Huibers, Wouter Faber","submitted_at":"2023-11-22T09:51:43Z","abstract_excerpt":"Generative Artificial Intelligence (AI) can be used to automatically generate medical reports based on transcripts of medical consultations. The aim is to reduce the administrative burden that healthcare professionals face. The accuracy of the generated reports needs to be established to ensure their correctness and usefulness. There are several metrics for measuring the accuracy of AI generated reports, but little work has been done towards the application of these metrics in medical reporting. A comparative experimentation of 10 accuracy metrics has been performed on AI generated medical rep"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.13273","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-22T09:51:43Z","cross_cats_sorted":[],"title_canon_sha256":"811467dfcea97d14f67806ade932c0be0a25b25b0e95c0d587c0733c9e91b3d2","abstract_canon_sha256":"4ea847b4d233915182169bc930dda1ce438333a3836b6264954511ec37f34e56"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:30:56.853179Z","signature_b64":"2htDD6nuePAXmELBh4zhoDLjJrBP2t5dmkS1Gf+Pz3kW9o6TGLKcpw+81MBhtyh3X6nVXwHjtoJs8G6PJrUKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c23718aaf12fe883f2757d3a111b1d459bf46b0529eace7eb105d9873fac689","last_reissued_at":"2026-07-05T07:30:56.852745Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:30:56.852745Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comparative Experimentation of Accuracy Metrics in Automated Medical Reporting: The Case of Otitis Consultations","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Renske Eline Bootsma, Sandra van Dulmen, Sjaak Brinkkemper, Tom Huibers, Wouter Faber","submitted_at":"2023-11-22T09:51:43Z","abstract_excerpt":"Generative Artificial Intelligence (AI) can be used to automatically generate medical reports based on transcripts of medical consultations. The aim is to reduce the administrative burden that healthcare professionals face. The accuracy of the generated reports needs to be established to ensure their correctness and usefulness. There are several metrics for measuring the accuracy of AI generated reports, but little work has been done towards the application of these metrics in medical reporting. A comparative experimentation of 10 accuracy metrics has been performed on AI generated medical rep"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.13273","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.13273/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.13273","created_at":"2026-07-05T07:30:56.852798+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.13273v2","created_at":"2026-07-05T07:30:56.852798+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.13273","created_at":"2026-07-05T07:30:56.852798+00:00"},{"alias_kind":"pith_short_12","alias_value":"DQRXDCVPCL7I","created_at":"2026-07-05T07:30:56.852798+00:00"},{"alias_kind":"pith_short_16","alias_value":"DQRXDCVPCL7IQPZH","created_at":"2026-07-05T07:30:56.852798+00:00"},{"alias_kind":"pith_short_8","alias_value":"DQRXDCVP","created_at":"2026-07-05T07:30:56.852798+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00095","citing_title":"ClinBench-HPB: A Clinical Benchmark for Evaluating LLMs in Hepato-Pancreato-Biliary Diseases","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R","json":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R.json","graph_json":"https://pith.science/api/pith-number/DQRXDCVPCL7IQPZHK7J2CENR2R/graph.json","events_json":"https://pith.science/api/pith-number/DQRXDCVPCL7IQPZHK7J2CENR2R/events.json","paper":"https://pith.science/paper/DQRXDCVP"},"agent_actions":{"view_html":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R","download_json":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R.json","view_paper":"https://pith.science/paper/DQRXDCVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.13273&json=true","fetch_graph":"https://pith.science/api/pith-number/DQRXDCVPCL7IQPZHK7J2CENR2R/graph.json","fetch_events":"https://pith.science/api/pith-number/DQRXDCVPCL7IQPZHK7J2CENR2R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R/action/storage_attestation","attest_author":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R/action/author_attestation","sign_citation":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R/action/citation_signature","submit_replication":"https://pith.science/pith/DQRXDCVPCL7IQPZHK7J2CENR2R/action/replication_record"}},"created_at":"2026-07-05T07:30:56.852798+00:00","updated_at":"2026-07-05T07:30:56.852798+00:00"}