{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:66JYSY4WYZKP6THYM7334VYMCO","short_pith_number":"pith:66JYSY4W","schema_version":"1.0","canonical_sha256":"f793896396c654ff4cf867f7be570c13964c44b23cda9798ec0fb57dcb87a928","source":{"kind":"arxiv","id":"2303.08448","version":1},"attestation_state":"computed","paper":{"title":"A Cross-institutional Evaluation on Breast Cancer Phenotyping NLP Algorithms on Electronic Health Records","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anne Blaes, Hongfang Liu, Ju Sun, Liwei Wang, Nan Wang, Rui Zhang, Sicheng Zhou","submitted_at":"2023-03-15T08:44:07Z","abstract_excerpt":"Objective: The generalizability of clinical large language models is usually ignored during the model development process. This study evaluated the generalizability of BERT-based clinical NLP models across different clinical settings through a breast cancer phenotype extraction task.\n  Materials and Methods: Two clinical corpora of breast cancer patients were collected from the electronic health records from the University of Minnesota and the Mayo Clinic, and annotated following the same guideline. We developed three types of NLP models (i.e., conditional random field, bi-directional long sho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.08448","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-15T08:44:07Z","cross_cats_sorted":["cs.IR","cs.LG"],"title_canon_sha256":"7dec0cfae97562b8946aefa16b30911e8d18e1474fb4836bb0ce7626c42f2038","abstract_canon_sha256":"f645e022f03595fefed839f76ccc4b396c509cb163190cb4dbe4f2348e774f97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:51:33.138411Z","signature_b64":"YarB43vwXpLC8GreBb1egBbzNCYIOxC06z63JVSPkZmY/Woet4hgJTBU5EvI5kx+JJwZZQYhUfWXm7OdST7UAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f793896396c654ff4cf867f7be570c13964c44b23cda9798ec0fb57dcb87a928","last_reissued_at":"2026-07-05T05:51:33.137888Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:51:33.137888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Cross-institutional Evaluation on Breast Cancer Phenotyping NLP Algorithms on Electronic Health Records","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anne Blaes, Hongfang Liu, Ju Sun, Liwei Wang, Nan Wang, Rui Zhang, Sicheng Zhou","submitted_at":"2023-03-15T08:44:07Z","abstract_excerpt":"Objective: The generalizability of clinical large language models is usually ignored during the model development process. This study evaluated the generalizability of BERT-based clinical NLP models across different clinical settings through a breast cancer phenotype extraction task.\n  Materials and Methods: Two clinical corpora of breast cancer patients were collected from the electronic health records from the University of Minnesota and the Mayo Clinic, and annotated following the same guideline. We developed three types of NLP models (i.e., conditional random field, bi-directional long sho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.08448","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.08448/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.08448","created_at":"2026-07-05T05:51:33.137953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.08448v1","created_at":"2026-07-05T05:51:33.137953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.08448","created_at":"2026-07-05T05:51:33.137953+00:00"},{"alias_kind":"pith_short_12","alias_value":"66JYSY4WYZKP","created_at":"2026-07-05T05:51:33.137953+00:00"},{"alias_kind":"pith_short_16","alias_value":"66JYSY4WYZKP6THY","created_at":"2026-07-05T05:51:33.137953+00:00"},{"alias_kind":"pith_short_8","alias_value":"66JYSY4W","created_at":"2026-07-05T05:51:33.137953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO","json":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO.json","graph_json":"https://pith.science/api/pith-number/66JYSY4WYZKP6THYM7334VYMCO/graph.json","events_json":"https://pith.science/api/pith-number/66JYSY4WYZKP6THYM7334VYMCO/events.json","paper":"https://pith.science/paper/66JYSY4W"},"agent_actions":{"view_html":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO","download_json":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO.json","view_paper":"https://pith.science/paper/66JYSY4W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.08448&json=true","fetch_graph":"https://pith.science/api/pith-number/66JYSY4WYZKP6THYM7334VYMCO/graph.json","fetch_events":"https://pith.science/api/pith-number/66JYSY4WYZKP6THYM7334VYMCO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO/action/storage_attestation","attest_author":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO/action/author_attestation","sign_citation":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO/action/citation_signature","submit_replication":"https://pith.science/pith/66JYSY4WYZKP6THYM7334VYMCO/action/replication_record"}},"created_at":"2026-07-05T05:51:33.137953+00:00","updated_at":"2026-07-05T05:51:33.137953+00:00"}