{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:64SUA3Z226JLTHBRB36AFVVD3A","short_pith_number":"pith:64SUA3Z2","schema_version":"1.0","canonical_sha256":"f725406f3ad792b99c310efc02d6a3d836bf9f5df5b227863711941dd62b40e2","source":{"kind":"arxiv","id":"2505.17167","version":1},"attestation_state":"computed","paper":{"title":"CRG Score: A Distribution-Aware Clinical Metric for Radiology Report Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Bernhard Kainz, Bjoern Menze, Hadrien Reynaud, Ibrahim Ethem Hamamci, Sezgin Er, Suprosanna Shit","submitted_at":"2025-05-22T17:02:28Z","abstract_excerpt":"Evaluating long-context radiology report generation is challenging. NLG metrics fail to capture clinical correctness, while LLM-based metrics often lack generalizability. Clinical accuracy metrics are more relevant but are sensitive to class imbalance, frequently favoring trivial predictions. We propose the CRG Score, a distribution-aware and adaptable metric that evaluates only clinically relevant abnormalities explicitly described in reference reports. CRG supports both binary and structured labels (e.g., type, location) and can be paired with any LLM for feature extraction. By balancing pen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17167","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-22T17:02:28Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"1d1ae9a88ea24649a2afdf3c486272a9701318c295d217289e0768d8a8eed695","abstract_canon_sha256":"f00fb73a7f134b9f2b298c175deb0cd98f2a6a9701dd913a5a5c479f86dd5697"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:48.228913Z","signature_b64":"8t3l7vc41p1ekVmO13EL3RA0lSxZVgurz8kxtLbGXvfKULYGKasuWxnB2ccDwGnXbRa8l20eFq0ARMOcvVF6DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f725406f3ad792b99c310efc02d6a3d836bf9f5df5b227863711941dd62b40e2","last_reissued_at":"2026-07-05T11:07:48.228407Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:48.228407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CRG Score: A Distribution-Aware Clinical Metric for Radiology Report Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Bernhard Kainz, Bjoern Menze, Hadrien Reynaud, Ibrahim Ethem Hamamci, Sezgin Er, Suprosanna Shit","submitted_at":"2025-05-22T17:02:28Z","abstract_excerpt":"Evaluating long-context radiology report generation is challenging. NLG metrics fail to capture clinical correctness, while LLM-based metrics often lack generalizability. Clinical accuracy metrics are more relevant but are sensitive to class imbalance, frequently favoring trivial predictions. We propose the CRG Score, a distribution-aware and adaptable metric that evaluates only clinically relevant abnormalities explicitly described in reference reports. CRG supports both binary and structured labels (e.g., type, location) and can be paired with any LLM for feature extraction. By balancing pen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17167","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17167/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17167","created_at":"2026-07-05T11:07:48.228457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17167v1","created_at":"2026-07-05T11:07:48.228457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17167","created_at":"2026-07-05T11:07:48.228457+00:00"},{"alias_kind":"pith_short_12","alias_value":"64SUA3Z226JL","created_at":"2026-07-05T11:07:48.228457+00:00"},{"alias_kind":"pith_short_16","alias_value":"64SUA3Z226JLTHBR","created_at":"2026-07-05T11:07:48.228457+00:00"},{"alias_kind":"pith_short_8","alias_value":"64SUA3Z2","created_at":"2026-07-05T11:07:48.228457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.10437","citing_title":"Enhancing Fine-Grained Spatial Grounding in 3D CT Report Generation via Discriminative Guidance","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A","json":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A.json","graph_json":"https://pith.science/api/pith-number/64SUA3Z226JLTHBRB36AFVVD3A/graph.json","events_json":"https://pith.science/api/pith-number/64SUA3Z226JLTHBRB36AFVVD3A/events.json","paper":"https://pith.science/paper/64SUA3Z2"},"agent_actions":{"view_html":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A","download_json":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A.json","view_paper":"https://pith.science/paper/64SUA3Z2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17167&json=true","fetch_graph":"https://pith.science/api/pith-number/64SUA3Z226JLTHBRB36AFVVD3A/graph.json","fetch_events":"https://pith.science/api/pith-number/64SUA3Z226JLTHBRB36AFVVD3A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A/action/storage_attestation","attest_author":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A/action/author_attestation","sign_citation":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A/action/citation_signature","submit_replication":"https://pith.science/pith/64SUA3Z226JLTHBRB36AFVVD3A/action/replication_record"}},"created_at":"2026-07-05T11:07:48.228457+00:00","updated_at":"2026-07-05T11:07:48.228457+00:00"}