{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6WCB253QMJLPVH4IVW2XIXXXCJ","short_pith_number":"pith:6WCB253Q","schema_version":"1.0","canonical_sha256":"f5841d77706256fa9f88adb5745ef71244f3c797f4f7ccddeeed32ccd5468db3","source":{"kind":"arxiv","id":"2202.05273","version":1},"attestation_state":"computed","paper":{"title":"Towards a Guideline for Evaluation Metrics in Medical Image Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"eess.IV","authors_text":"Dominik M\\\"uller, Frank Kramer, I\\~naki Soto-Rey","submitted_at":"2022-02-10T13:38:05Z","abstract_excerpt":"In the last decade, research on artificial intelligence has seen rapid growth with deep learning models, especially in the field of medical image segmentation. Various studies demonstrated that these models have powerful prediction capabilities and achieved similar results as clinicians. However, recent studies revealed that the evaluation in image segmentation studies lacks reliable model performance assessment and showed statistical bias by incorrect metric implementation or usage. Thus, this work provides an overview and interpretation guide on the following metrics for medical image segmen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.05273","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.IV","submitted_at":"2022-02-10T13:38:05Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"9b9bf0027f7f4e7af7ac2d81a918bcb1827ed0b7aeeeeb6fa96c4bdcfffdc4a0","abstract_canon_sha256":"197c86ec0caf130f27c6f5453daa52a46bfc48513d85fd52369d5740d21f000a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:56:06.717454Z","signature_b64":"PuFVFJakP8Lz47mD/rKs9cZ838hoKIe96EAZvucl0JRMTcPmP+yAC3Km2I/ypOyqjc0nY1UmzDWTjq7vdS+lBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5841d77706256fa9f88adb5745ef71244f3c797f4f7ccddeeed32ccd5468db3","last_reissued_at":"2026-07-05T03:56:06.717100Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:56:06.717100Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards a Guideline for Evaluation Metrics in Medical Image Segmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"eess.IV","authors_text":"Dominik M\\\"uller, Frank Kramer, I\\~naki Soto-Rey","submitted_at":"2022-02-10T13:38:05Z","abstract_excerpt":"In the last decade, research on artificial intelligence has seen rapid growth with deep learning models, especially in the field of medical image segmentation. Various studies demonstrated that these models have powerful prediction capabilities and achieved similar results as clinicians. However, recent studies revealed that the evaluation in image segmentation studies lacks reliable model performance assessment and showed statistical bias by incorrect metric implementation or usage. Thus, this work provides an overview and interpretation guide on the following metrics for medical image segmen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.05273","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.05273/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.05273","created_at":"2026-07-05T03:56:06.717157+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.05273v1","created_at":"2026-07-05T03:56:06.717157+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.05273","created_at":"2026-07-05T03:56:06.717157+00:00"},{"alias_kind":"pith_short_12","alias_value":"6WCB253QMJLP","created_at":"2026-07-05T03:56:06.717157+00:00"},{"alias_kind":"pith_short_16","alias_value":"6WCB253QMJLPVH4I","created_at":"2026-07-05T03:56:06.717157+00:00"},{"alias_kind":"pith_short_8","alias_value":"6WCB253Q","created_at":"2026-07-05T03:56:06.717157+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.06327","citing_title":"Can Diffusion Models Bridge the Domain Gap in Cardiac MR Imaging?","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ","json":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ.json","graph_json":"https://pith.science/api/pith-number/6WCB253QMJLPVH4IVW2XIXXXCJ/graph.json","events_json":"https://pith.science/api/pith-number/6WCB253QMJLPVH4IVW2XIXXXCJ/events.json","paper":"https://pith.science/paper/6WCB253Q"},"agent_actions":{"view_html":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ","download_json":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ.json","view_paper":"https://pith.science/paper/6WCB253Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.05273&json=true","fetch_graph":"https://pith.science/api/pith-number/6WCB253QMJLPVH4IVW2XIXXXCJ/graph.json","fetch_events":"https://pith.science/api/pith-number/6WCB253QMJLPVH4IVW2XIXXXCJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ/action/storage_attestation","attest_author":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ/action/author_attestation","sign_citation":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ/action/citation_signature","submit_replication":"https://pith.science/pith/6WCB253QMJLPVH4IVW2XIXXXCJ/action/replication_record"}},"created_at":"2026-07-05T03:56:06.717157+00:00","updated_at":"2026-07-05T03:56:06.717157+00:00"}