{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5JRCPKPN6KLROTSMPW7CA5SSEU","short_pith_number":"pith:5JRCPKPN","schema_version":"1.0","canonical_sha256":"ea6227a9edf297174e4c7dbe207652251de0b16913824a670b98057209fbff5a","source":{"kind":"arxiv","id":"2301.12473","version":2},"attestation_state":"computed","paper":{"title":"Large Language Models for Biomedical Knowledge Graph Construction: Information extraction from EMR notes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Davit Shahnazaryan, Fadi Shaya, Kent Small, Spartak Bughdaryan, Vahan Arsenyan","submitted_at":"2023-01-29T15:52:33Z","abstract_excerpt":"The automatic construction of knowledge graphs (KGs) is an important research area in medicine, with far-reaching applications spanning drug discovery and clinical trial design. These applications hinge on the accurate identification of interactions among medical and biological entities. In this study, we propose an end-to-end machine learning solution based on large language models (LLMs) that utilize electronic medical record notes to construct KGs. The entities used in the KG construction process are diseases, factors, treatments, as well as manifestations that coexist with the patient whil"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.12473","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-01-29T15:52:33Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"626c152877c00aea7ceadc012b910a4332ff802f22e4d04f4068a694e7c78d7d","abstract_canon_sha256":"201e4bfb9161d0deb15fcd2566d4d4b51e2ab7a7cfc5d0a7dd70b608a6d40822"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:36.905203Z","signature_b64":"4uP58iAeJf0KjzmOJj7UuKLyn5HA4StcEvY7vVz+OO/ryf8w5lYw0bkyYx/MEBK/qY2Eq3H9ys/0miKU1+UeAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea6227a9edf297174e4c7dbe207652251de0b16913824a670b98057209fbff5a","last_reissued_at":"2026-07-05T10:06:36.904771Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:36.904771Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Biomedical Knowledge Graph Construction: Information extraction from EMR notes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Davit Shahnazaryan, Fadi Shaya, Kent Small, Spartak Bughdaryan, Vahan Arsenyan","submitted_at":"2023-01-29T15:52:33Z","abstract_excerpt":"The automatic construction of knowledge graphs (KGs) is an important research area in medicine, with far-reaching applications spanning drug discovery and clinical trial design. These applications hinge on the accurate identification of interactions among medical and biological entities. In this study, we propose an end-to-end machine learning solution based on large language models (LLMs) that utilize electronic medical record notes to construct KGs. The entities used in the KG construction process are diseases, factors, treatments, as well as manifestations that coexist with the patient whil"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.12473","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.12473/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.12473","created_at":"2026-07-05T10:06:36.904837+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.12473v2","created_at":"2026-07-05T10:06:36.904837+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.12473","created_at":"2026-07-05T10:06:36.904837+00:00"},{"alias_kind":"pith_short_12","alias_value":"5JRCPKPN6KLR","created_at":"2026-07-05T10:06:36.904837+00:00"},{"alias_kind":"pith_short_16","alias_value":"5JRCPKPN6KLROTSM","created_at":"2026-07-05T10:06:36.904837+00:00"},{"alias_kind":"pith_short_8","alias_value":"5JRCPKPN","created_at":"2026-07-05T10:06:36.904837+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.15256","citing_title":"Structured Extraction of Real World Medical Knowledge using LLMs for Summarization and Search","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU","json":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU.json","graph_json":"https://pith.science/api/pith-number/5JRCPKPN6KLROTSMPW7CA5SSEU/graph.json","events_json":"https://pith.science/api/pith-number/5JRCPKPN6KLROTSMPW7CA5SSEU/events.json","paper":"https://pith.science/paper/5JRCPKPN"},"agent_actions":{"view_html":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU","download_json":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU.json","view_paper":"https://pith.science/paper/5JRCPKPN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.12473&json=true","fetch_graph":"https://pith.science/api/pith-number/5JRCPKPN6KLROTSMPW7CA5SSEU/graph.json","fetch_events":"https://pith.science/api/pith-number/5JRCPKPN6KLROTSMPW7CA5SSEU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU/action/storage_attestation","attest_author":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU/action/author_attestation","sign_citation":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU/action/citation_signature","submit_replication":"https://pith.science/pith/5JRCPKPN6KLROTSMPW7CA5SSEU/action/replication_record"}},"created_at":"2026-07-05T10:06:36.904837+00:00","updated_at":"2026-07-05T10:06:36.904837+00:00"}