{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HJQ2H2PYKDLJE5W3TV3HMV7C3W","short_pith_number":"pith:HJQ2H2PY","schema_version":"1.0","canonical_sha256":"3a61a3e9f850d69276db9d767657e2dda12657936d63410d7db7c0e5809e2de6","source":{"kind":"arxiv","id":"2503.04930","version":1},"attestation_state":"computed","paper":{"title":"HILGEN: Hierarchically-Informed Data Generation for Biomedical NER Using Knowledgebases and Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abeed Sarker, Selen Bozkurt, Sudeshna Das, Swati Rajwal, Yao Ge, Yuting Guo","submitted_at":"2025-03-06T20:02:19Z","abstract_excerpt":"We present HILGEN, a Hierarchically-Informed Data Generation approach that combines domain knowledge from the Unified Medical Language System (UMLS) with synthetic data generated by large language models (LLMs), specifically GPT-3.5. Our approach leverages UMLS's hierarchical structure to expand training data with related concepts, while incorporating contextual information from LLMs through targeted prompts aimed at automatically generating synthetic examples for sparsely occurring named entities. The performance of the HILGEN approach was evaluated across four biomedical NER datasets (MIMIC "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.04930","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-06T20:02:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fed20fd22d8f5207d11bffa4cc192e889ea9ece553e68e75e349603f79b77336","abstract_canon_sha256":"477bdaf7b494cce03e169999a897b3711bb8b2314c8356005d83f78fec949643"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:51.706619Z","signature_b64":"LcvNmv9S1sqHqkUOpF5RkXA8cp8R5P4CK3yxAFsxyxCBCHadDGSyAkzsy9W7gy5AlgwmEH5nCsKTKq/ZNGDFBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a61a3e9f850d69276db9d767657e2dda12657936d63410d7db7c0e5809e2de6","last_reissued_at":"2026-07-05T10:25:51.705590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:51.705590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HILGEN: Hierarchically-Informed Data Generation for Biomedical NER Using Knowledgebases and Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Abeed Sarker, Selen Bozkurt, Sudeshna Das, Swati Rajwal, Yao Ge, Yuting Guo","submitted_at":"2025-03-06T20:02:19Z","abstract_excerpt":"We present HILGEN, a Hierarchically-Informed Data Generation approach that combines domain knowledge from the Unified Medical Language System (UMLS) with synthetic data generated by large language models (LLMs), specifically GPT-3.5. Our approach leverages UMLS's hierarchical structure to expand training data with related concepts, while incorporating contextual information from LLMs through targeted prompts aimed at automatically generating synthetic examples for sparsely occurring named entities. The performance of the HILGEN approach was evaluated across four biomedical NER datasets (MIMIC "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.04930","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.04930/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.04930","created_at":"2026-07-05T10:25:51.705753+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.04930v1","created_at":"2026-07-05T10:25:51.705753+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.04930","created_at":"2026-07-05T10:25:51.705753+00:00"},{"alias_kind":"pith_short_12","alias_value":"HJQ2H2PYKDLJ","created_at":"2026-07-05T10:25:51.705753+00:00"},{"alias_kind":"pith_short_16","alias_value":"HJQ2H2PYKDLJE5W3","created_at":"2026-07-05T10:25:51.705753+00:00"},{"alias_kind":"pith_short_8","alias_value":"HJQ2H2PY","created_at":"2026-07-05T10:25:51.705753+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W","json":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W.json","graph_json":"https://pith.science/api/pith-number/HJQ2H2PYKDLJE5W3TV3HMV7C3W/graph.json","events_json":"https://pith.science/api/pith-number/HJQ2H2PYKDLJE5W3TV3HMV7C3W/events.json","paper":"https://pith.science/paper/HJQ2H2PY"},"agent_actions":{"view_html":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W","download_json":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W.json","view_paper":"https://pith.science/paper/HJQ2H2PY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.04930&json=true","fetch_graph":"https://pith.science/api/pith-number/HJQ2H2PYKDLJE5W3TV3HMV7C3W/graph.json","fetch_events":"https://pith.science/api/pith-number/HJQ2H2PYKDLJE5W3TV3HMV7C3W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W/action/storage_attestation","attest_author":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W/action/author_attestation","sign_citation":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W/action/citation_signature","submit_replication":"https://pith.science/pith/HJQ2H2PYKDLJE5W3TV3HMV7C3W/action/replication_record"}},"created_at":"2026-07-05T10:25:51.705753+00:00","updated_at":"2026-07-05T10:25:51.705753+00:00"}