{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RL6XWOQ332F7KJW5E637IBE4V2","short_pith_number":"pith:RL6XWOQ3","schema_version":"1.0","canonical_sha256":"8afd7b3a1bde8bf526dd27b7f4049cae8de24ae47e8d58b89b8c530ea59f109a","source":{"kind":"arxiv","id":"2412.17803","version":2},"attestation_state":"computed","paper":{"title":"Examining Imbalance Effects on Performance and Demographic Fairness of Clinical Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"I-Chan Huang, Precious Jones, Weisi Liu, Xiaolei Huang","submitted_at":"2024-12-23T18:58:11Z","abstract_excerpt":"Data imbalance is a fundamental challenge in applying language models to biomedical applications, particularly in ICD code prediction tasks where label and demographic distributions are uneven. While state-of-the-art language models have been increasingly adopted in biomedical tasks, few studies have systematically examined how data imbalance affects model performance and fairness across demographic groups. This study fills the gap by statistically probing the relationship between data imbalance and model performance in ICD code prediction. We analyze imbalances in a standard benchmark data ac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17803","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-23T18:58:11Z","cross_cats_sorted":[],"title_canon_sha256":"47cb96dfc2cf4fc3b04d5094d0186d475df058a970674f50de93aca1b94e27f2","abstract_canon_sha256":"3fea9fbe66bc3cd605fafa3f6a640c9782a63249568e381e549e8419213ea994"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:07.953730Z","signature_b64":"U/3ZsTu8Myw1lS7hh5+k8PhsToI1VV/ek46JXeOxyyOOm6f2JFXioINJErHUha6qxqjo5o966qE2dpeH+KQ9CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8afd7b3a1bde8bf526dd27b7f4049cae8de24ae47e8d58b89b8c530ea59f109a","last_reissued_at":"2026-07-05T10:14:07.953227Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:07.953227Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Examining Imbalance Effects on Performance and Demographic Fairness of Clinical Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"I-Chan Huang, Precious Jones, Weisi Liu, Xiaolei Huang","submitted_at":"2024-12-23T18:58:11Z","abstract_excerpt":"Data imbalance is a fundamental challenge in applying language models to biomedical applications, particularly in ICD code prediction tasks where label and demographic distributions are uneven. While state-of-the-art language models have been increasingly adopted in biomedical tasks, few studies have systematically examined how data imbalance affects model performance and fairness across demographic groups. This study fills the gap by statistically probing the relationship between data imbalance and model performance in ICD code prediction. We analyze imbalances in a standard benchmark data ac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17803","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17803/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17803","created_at":"2026-07-05T10:14:07.953287+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17803v2","created_at":"2026-07-05T10:14:07.953287+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17803","created_at":"2026-07-05T10:14:07.953287+00:00"},{"alias_kind":"pith_short_12","alias_value":"RL6XWOQ332F7","created_at":"2026-07-05T10:14:07.953287+00:00"},{"alias_kind":"pith_short_16","alias_value":"RL6XWOQ332F7KJW5","created_at":"2026-07-05T10:14:07.953287+00:00"},{"alias_kind":"pith_short_8","alias_value":"RL6XWOQ3","created_at":"2026-07-05T10:14:07.953287+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.03498","citing_title":"Resource-Conscious Modeling for Next- Day Discharge Prediction Using Clinical Notes","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2","json":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2.json","graph_json":"https://pith.science/api/pith-number/RL6XWOQ332F7KJW5E637IBE4V2/graph.json","events_json":"https://pith.science/api/pith-number/RL6XWOQ332F7KJW5E637IBE4V2/events.json","paper":"https://pith.science/paper/RL6XWOQ3"},"agent_actions":{"view_html":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2","download_json":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2.json","view_paper":"https://pith.science/paper/RL6XWOQ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17803&json=true","fetch_graph":"https://pith.science/api/pith-number/RL6XWOQ332F7KJW5E637IBE4V2/graph.json","fetch_events":"https://pith.science/api/pith-number/RL6XWOQ332F7KJW5E637IBE4V2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2/action/storage_attestation","attest_author":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2/action/author_attestation","sign_citation":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2/action/citation_signature","submit_replication":"https://pith.science/pith/RL6XWOQ332F7KJW5E637IBE4V2/action/replication_record"}},"created_at":"2026-07-05T10:14:07.953287+00:00","updated_at":"2026-07-05T10:14:07.953287+00:00"}