{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:EALTS4T4XJ5YH2B3YQ2CKNNHET","short_pith_number":"pith:EALTS4T4","schema_version":"1.0","canonical_sha256":"201739727cba7b83e83bc4342535a724cb7d8da47cbbe4115073d7192857bd3c","source":{"kind":"arxiv","id":"2211.09455","version":1},"attestation_state":"computed","paper":{"title":"Consultation Checklists: Standardising the Human Evaluation of Medical Note Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aleksandar Savkov, Alex Papadopoulos Korfiatis, Anya Belz, Ehud Reiter, Francesco Moramarco, Mark Perera","submitted_at":"2022-11-17T10:54:28Z","abstract_excerpt":"Evaluating automatically generated text is generally hard due to the inherently subjective nature of many aspects of the output quality. This difficulty is compounded in automatic consultation note generation by differing opinions between medical experts both about which patient statements should be included in generated notes and about their respective importance in arriving at a diagnosis. Previous real-world evaluations of note-generation systems saw substantial disagreement between expert evaluators. In this paper we propose a protocol that aims to increase objectivity by grounding evaluat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.09455","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-11-17T10:54:28Z","cross_cats_sorted":[],"title_canon_sha256":"2c31d864a38ed6e7b5703e49b2322b52ccc0ff6ae0839a242257075d4b9efb53","abstract_canon_sha256":"3e7153dd0835a479323f65e184a667dead65b2b3bbabf3f7d8723c675a0d1f4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:16:56.053046Z","signature_b64":"6+GecWeadgGxselLBkGp4XQ2Pd1qqoLzpPebaW05ErZUxYRB7ejClpt3GL+uorg9RSN6lUkhaJFB+9yKr1fbDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"201739727cba7b83e83bc4342535a724cb7d8da47cbbe4115073d7192857bd3c","last_reissued_at":"2026-07-05T05:16:56.052651Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:16:56.052651Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Consultation Checklists: Standardising the Human Evaluation of Medical Note Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aleksandar Savkov, Alex Papadopoulos Korfiatis, Anya Belz, Ehud Reiter, Francesco Moramarco, Mark Perera","submitted_at":"2022-11-17T10:54:28Z","abstract_excerpt":"Evaluating automatically generated text is generally hard due to the inherently subjective nature of many aspects of the output quality. This difficulty is compounded in automatic consultation note generation by differing opinions between medical experts both about which patient statements should be included in generated notes and about their respective importance in arriving at a diagnosis. Previous real-world evaluations of note-generation systems saw substantial disagreement between expert evaluators. In this paper we propose a protocol that aims to increase objectivity by grounding evaluat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.09455","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.09455/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.09455","created_at":"2026-07-05T05:16:56.052708+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.09455v1","created_at":"2026-07-05T05:16:56.052708+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.09455","created_at":"2026-07-05T05:16:56.052708+00:00"},{"alias_kind":"pith_short_12","alias_value":"EALTS4T4XJ5Y","created_at":"2026-07-05T05:16:56.052708+00:00"},{"alias_kind":"pith_short_16","alias_value":"EALTS4T4XJ5YH2B3","created_at":"2026-07-05T05:16:56.052708+00:00"},{"alias_kind":"pith_short_8","alias_value":"EALTS4T4","created_at":"2026-07-05T05:16:56.052708+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.04254","citing_title":"CLINICSUM: Utilizing Language Models for Generating Clinical Summaries from Patient-Doctor Conversations","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET","json":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET.json","graph_json":"https://pith.science/api/pith-number/EALTS4T4XJ5YH2B3YQ2CKNNHET/graph.json","events_json":"https://pith.science/api/pith-number/EALTS4T4XJ5YH2B3YQ2CKNNHET/events.json","paper":"https://pith.science/paper/EALTS4T4"},"agent_actions":{"view_html":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET","download_json":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET.json","view_paper":"https://pith.science/paper/EALTS4T4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.09455&json=true","fetch_graph":"https://pith.science/api/pith-number/EALTS4T4XJ5YH2B3YQ2CKNNHET/graph.json","fetch_events":"https://pith.science/api/pith-number/EALTS4T4XJ5YH2B3YQ2CKNNHET/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET/action/storage_attestation","attest_author":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET/action/author_attestation","sign_citation":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET/action/citation_signature","submit_replication":"https://pith.science/pith/EALTS4T4XJ5YH2B3YQ2CKNNHET/action/replication_record"}},"created_at":"2026-07-05T05:16:56.052708+00:00","updated_at":"2026-07-05T05:16:56.052708+00:00"}