{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GRPVI3DUZ7JPEZFZFP3Q3KWWC3","short_pith_number":"pith:GRPVI3DU","schema_version":"1.0","canonical_sha256":"345f546c74cfd2f264b92bf70daad616e3229f475cd398449e8628889bb2fde9","source":{"kind":"arxiv","id":"2503.05347","version":2},"attestation_state":"computed","paper":{"title":"GEMA-Score: Granular Explainable Multi-Agent Scoring Framework for Radiology Report Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.CL","authors_text":"Dominic C Marshall, Guang Yang, Huichi Zhou, Jiahao Huang, Kinhei Lee, Peiyuan Jing, Weihang Deng, Yingying Fang, Zhenxuan Zhang, Zhifan Gao, Zihao Jin","submitted_at":"2025-03-07T11:42:22Z","abstract_excerpt":"Automatic medical report generation has the potential to support clinical diagnosis, reduce the workload of radiologists, and demonstrate potential for enhancing diagnostic consistency. However, current evaluation metrics often fail to reflect the clinical reliability of generated reports. Early overlap-based methods focus on textual matches between predicted and ground-truth entities but miss fine-grained clinical details (e.g., anatomical location, severity). Some diagnostic metrics are limited by fixed vocabularies or templates, reducing their ability to capture diverse clinical expressions"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.05347","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-03-07T11:42:22Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"25bb3e498b7f897609cedd7004425ab36f80a8e7af14286ef1d3f3d4df044b76","abstract_canon_sha256":"c77aefb07dd7856d88ad220dadce50d5499156652db4673096890342393cc765"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:23.458037Z","signature_b64":"er1aY19Nyad76K/zv3FiDEVbTStyXbpO0TK4kKiFCpOiju7sv2qdEKlgmhQorgihRyC/sii8LGMqBGB/oCVKBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"345f546c74cfd2f264b92bf70daad616e3229f475cd398449e8628889bb2fde9","last_reissued_at":"2026-07-05T11:48:23.457528Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:23.457528Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GEMA-Score: Granular Explainable Multi-Agent Scoring Framework for Radiology Report Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.CL","authors_text":"Dominic C Marshall, Guang Yang, Huichi Zhou, Jiahao Huang, Kinhei Lee, Peiyuan Jing, Weihang Deng, Yingying Fang, Zhenxuan Zhang, Zhifan Gao, Zihao Jin","submitted_at":"2025-03-07T11:42:22Z","abstract_excerpt":"Automatic medical report generation has the potential to support clinical diagnosis, reduce the workload of radiologists, and demonstrate potential for enhancing diagnostic consistency. However, current evaluation metrics often fail to reflect the clinical reliability of generated reports. Early overlap-based methods focus on textual matches between predicted and ground-truth entities but miss fine-grained clinical details (e.g., anatomical location, severity). Some diagnostic metrics are limited by fixed vocabularies or templates, reducing their ability to capture diverse clinical expressions"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.05347","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.05347/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.05347","created_at":"2026-07-05T11:48:23.457587+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.05347v2","created_at":"2026-07-05T11:48:23.457587+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.05347","created_at":"2026-07-05T11:48:23.457587+00:00"},{"alias_kind":"pith_short_12","alias_value":"GRPVI3DUZ7JP","created_at":"2026-07-05T11:48:23.457587+00:00"},{"alias_kind":"pith_short_16","alias_value":"GRPVI3DUZ7JPEZFZ","created_at":"2026-07-05T11:48:23.457587+00:00"},{"alias_kind":"pith_short_8","alias_value":"GRPVI3DU","created_at":"2026-07-05T11:48:23.457587+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.26959","citing_title":"CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3","json":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3.json","graph_json":"https://pith.science/api/pith-number/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/graph.json","events_json":"https://pith.science/api/pith-number/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/events.json","paper":"https://pith.science/paper/GRPVI3DU"},"agent_actions":{"view_html":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3","download_json":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3.json","view_paper":"https://pith.science/paper/GRPVI3DU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.05347&json=true","fetch_graph":"https://pith.science/api/pith-number/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/graph.json","fetch_events":"https://pith.science/api/pith-number/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/action/storage_attestation","attest_author":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/action/author_attestation","sign_citation":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/action/citation_signature","submit_replication":"https://pith.science/pith/GRPVI3DUZ7JPEZFZFP3Q3KWWC3/action/replication_record"}},"created_at":"2026-07-05T11:48:23.457587+00:00","updated_at":"2026-07-05T11:48:23.457587+00:00"}