{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Z4AHT6JA73PCS46HUVTGNQWGLY","short_pith_number":"pith:Z4AHT6JA","schema_version":"1.0","canonical_sha256":"cf0079f920fede2973c7a56666c2c65e2203c7b04ce1b0758e51c345e0b69899","source":{"kind":"arxiv","id":"2311.17107","version":1},"attestation_state":"computed","paper":{"title":"ClimateX: Do LLMs Accurately Assess Human Expert Confidence in Climate Statements?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.IR"],"primary_cat":"cs.LG","authors_text":"Eddie Dilworth, Kerrie Wu, Romain Lacombe","submitted_at":"2023-11-28T10:26:57Z","abstract_excerpt":"Evaluating the accuracy of outputs generated by Large Language Models (LLMs) is especially important in the climate science and policy domain. We introduce the Expert Confidence in Climate Statements (ClimateX) dataset, a novel, curated, expert-labeled dataset consisting of 8094 climate statements collected from the latest Intergovernmental Panel on Climate Change (IPCC) reports, labeled with their associated confidence levels. Using this dataset, we show that recent LLMs can classify human expert confidence in climate-related statements, especially in a few-shot learning setting, but with lim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.17107","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-28T10:26:57Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY","cs.IR"],"title_canon_sha256":"5b7f6b3bc65d8d5dfe495ea1d37b297d5d244d212a5fa501ae95447c02cd026e","abstract_canon_sha256":"64cca5d188c6667420fb828b4b412e8bd7ee3158a2977708b2f50750a728494c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:57.569345Z","signature_b64":"NQi/NADxlaYot2+nJejDEBZGxq2x1ElTeferHWmR3RrKni1RFDaAR6H3TumLIUvMDFSGANPtloM0SlRm2DzpAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf0079f920fede2973c7a56666c2c65e2203c7b04ce1b0758e51c345e0b69899","last_reissued_at":"2026-07-05T07:17:57.568857Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:57.568857Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ClimateX: Do LLMs Accurately Assess Human Expert Confidence in Climate Statements?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.IR"],"primary_cat":"cs.LG","authors_text":"Eddie Dilworth, Kerrie Wu, Romain Lacombe","submitted_at":"2023-11-28T10:26:57Z","abstract_excerpt":"Evaluating the accuracy of outputs generated by Large Language Models (LLMs) is especially important in the climate science and policy domain. We introduce the Expert Confidence in Climate Statements (ClimateX) dataset, a novel, curated, expert-labeled dataset consisting of 8094 climate statements collected from the latest Intergovernmental Panel on Climate Change (IPCC) reports, labeled with their associated confidence levels. Using this dataset, we show that recent LLMs can classify human expert confidence in climate-related statements, especially in a few-shot learning setting, but with lim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.17107","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.17107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.17107","created_at":"2026-07-05T07:17:57.568920+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.17107v1","created_at":"2026-07-05T07:17:57.568920+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.17107","created_at":"2026-07-05T07:17:57.568920+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z4AHT6JA73PC","created_at":"2026-07-05T07:17:57.568920+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z4AHT6JA73PCS46H","created_at":"2026-07-05T07:17:57.568920+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z4AHT6JA","created_at":"2026-07-05T07:17:57.568920+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.13802","citing_title":"Enhancing LLMs for Governance with Human Oversight: Evaluating and Aligning LLMs on Expert Classification of Climate Misinformation for Detecting False or Misleading Claims about Climate Change","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY","json":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY.json","graph_json":"https://pith.science/api/pith-number/Z4AHT6JA73PCS46HUVTGNQWGLY/graph.json","events_json":"https://pith.science/api/pith-number/Z4AHT6JA73PCS46HUVTGNQWGLY/events.json","paper":"https://pith.science/paper/Z4AHT6JA"},"agent_actions":{"view_html":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY","download_json":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY.json","view_paper":"https://pith.science/paper/Z4AHT6JA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.17107&json=true","fetch_graph":"https://pith.science/api/pith-number/Z4AHT6JA73PCS46HUVTGNQWGLY/graph.json","fetch_events":"https://pith.science/api/pith-number/Z4AHT6JA73PCS46HUVTGNQWGLY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY/action/storage_attestation","attest_author":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY/action/author_attestation","sign_citation":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY/action/citation_signature","submit_replication":"https://pith.science/pith/Z4AHT6JA73PCS46HUVTGNQWGLY/action/replication_record"}},"created_at":"2026-07-05T07:17:57.568920+00:00","updated_at":"2026-07-05T07:17:57.568920+00:00"}