{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6KKY5CM3FSL7BFVLHETLK3UVQ6","short_pith_number":"pith:6KKY5CM3","schema_version":"1.0","canonical_sha256":"f2958e899b2c97f096ab3926b56e95878104a7a558f8ddd5060a0264b00b83f3","source":{"kind":"arxiv","id":"2205.12689","version":2},"attestation_state":"computed","paper":{"title":"Large Language Models are Few-Shot Clinical Information Extractors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"David Sontag, Hunter Lang, Monica Agrawal, Stefan Hegselmann, Yoon Kim","submitted_at":"2022-05-25T11:49:58Z","abstract_excerpt":"A long-running goal of the clinical NLP community is the extraction of important variables trapped in clinical notes. However, roadblocks have included dataset shift from the general domain and a lack of public clinical corpora and annotations. In this work, we show that large language models, such as InstructGPT, perform well at zero- and few-shot information extraction from clinical text despite not being trained specifically for the clinical domain. Whereas text classification and generation performance have already been studied extensively in such models, here we additionally demonstrate h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.12689","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-05-25T11:49:58Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"be13c61ce8d8f94c6a4f3e7f5a849f34a11a58af1b1f02936e1b54c65f561b64","abstract_canon_sha256":"3f21df9d1a72a7ef224fff4e188db3880826144b3a9bdfa5dc8616d82c91e3a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:21:01.361617Z","signature_b64":"JL9bEiVufcpyvPkUxSofcoLRcu0BCX75liJmmwayJuM88OmeiQ76J4h88QHq9MOC2HV3VaHo2tyeVXV8l9EQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2958e899b2c97f096ab3926b56e95878104a7a558f8ddd5060a0264b00b83f3","last_reissued_at":"2026-07-05T05:21:01.361124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:21:01.361124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models are Few-Shot Clinical Information Extractors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"David Sontag, Hunter Lang, Monica Agrawal, Stefan Hegselmann, Yoon Kim","submitted_at":"2022-05-25T11:49:58Z","abstract_excerpt":"A long-running goal of the clinical NLP community is the extraction of important variables trapped in clinical notes. However, roadblocks have included dataset shift from the general domain and a lack of public clinical corpora and annotations. In this work, we show that large language models, such as InstructGPT, perform well at zero- and few-shot information extraction from clinical text despite not being trained specifically for the clinical domain. Whereas text classification and generation performance have already been studied extensively in such models, here we additionally demonstrate h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.12689","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.12689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.12689","created_at":"2026-07-05T05:21:01.361185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.12689v2","created_at":"2026-07-05T05:21:01.361185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.12689","created_at":"2026-07-05T05:21:01.361185+00:00"},{"alias_kind":"pith_short_12","alias_value":"6KKY5CM3FSL7","created_at":"2026-07-05T05:21:01.361185+00:00"},{"alias_kind":"pith_short_16","alias_value":"6KKY5CM3FSL7BFVL","created_at":"2026-07-05T05:21:01.361185+00:00"},{"alias_kind":"pith_short_8","alias_value":"6KKY5CM3","created_at":"2026-07-05T05:21:01.361185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07761","citing_title":"Aligning Clinical Needs and AI Capabilities: A Survey on LLMs for Medical Reasoning","ref_index":47,"is_internal_anchor":true},{"citing_arxiv_id":"2305.09617","citing_title":"Towards Expert-Level Medical Question Answering with Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05740","citing_title":"RECOVER: Designing a Large Language Model-based Remote Patient Monitoring System for Postoperative Gastrointestinal Cancer Care","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.07738","citing_title":"Automated Construction of a Knowledge Graph of Nuclear Fusion Energy for Effective Elicitation and Retrieval of Information","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06028","citing_title":"A Multi-Stage Validation Framework for Trustworthy Large-scale Clinical Information Extraction using Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15547","citing_title":"Consistency Analysis of Sentiment Predictions using Syntactic & Semantic Context Assessment Summarization (SSAS)","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6","json":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6.json","graph_json":"https://pith.science/api/pith-number/6KKY5CM3FSL7BFVLHETLK3UVQ6/graph.json","events_json":"https://pith.science/api/pith-number/6KKY5CM3FSL7BFVLHETLK3UVQ6/events.json","paper":"https://pith.science/paper/6KKY5CM3"},"agent_actions":{"view_html":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6","download_json":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6.json","view_paper":"https://pith.science/paper/6KKY5CM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.12689&json=true","fetch_graph":"https://pith.science/api/pith-number/6KKY5CM3FSL7BFVLHETLK3UVQ6/graph.json","fetch_events":"https://pith.science/api/pith-number/6KKY5CM3FSL7BFVLHETLK3UVQ6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6/action/storage_attestation","attest_author":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6/action/author_attestation","sign_citation":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6/action/citation_signature","submit_replication":"https://pith.science/pith/6KKY5CM3FSL7BFVLHETLK3UVQ6/action/replication_record"}},"created_at":"2026-07-05T05:21:01.361185+00:00","updated_at":"2026-07-05T05:21:01.361185+00:00"}