{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5KJX2UVA7WVKP2OQ276QQOGY7K","short_pith_number":"pith:5KJX2UVA","schema_version":"1.0","canonical_sha256":"ea937d52a0fdaaa7e9d0d7fd0838d8fa8f3d5b8bbd8acaa24393ce4205833de6","source":{"kind":"arxiv","id":"2401.06730","version":2},"attestation_state":"computed","paper":{"title":"Relying on the Unreliable: The Impact of Language Models' Reluctance to Express Uncertainty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Jena D. Hwang, Kaitlyn Zhou, Maarten Sap, Xiang Ren","submitted_at":"2024-01-12T18:03:30Z","abstract_excerpt":"As natural language becomes the default interface for human-AI interaction, there is a need for LMs to appropriately communicate uncertainties in downstream applications. In this work, we investigate how LMs incorporate confidence in responses via natural language and how downstream users behave in response to LM-articulated uncertainties. We examine publicly deployed models and find that LMs are reluctant to express uncertainties when answering questions even when they produce incorrect responses. LMs can be explicitly prompted to express confidences, but tend to be overconfident, resulting i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06730","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-12T18:03:30Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"da5a1373e803fedd66e70544eda8722b7e41caae425e20573a8a837ebaef782d","abstract_canon_sha256":"06fdeb3d5986ec66d1c62eeb3d2baf70983341b81a1f84c89b0ccf43cb608f71"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:42:01.305343Z","signature_b64":"+Zmx7zeJwfPy34qq8/qfbwC3mo1wD3TNJNi6PXVR50PYVg+B+/Yj/pl3HOKipTZJclonPtPRWskxRKiycCDrDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea937d52a0fdaaa7e9d0d7fd0838d8fa8f3d5b8bbd8acaa24393ce4205833de6","last_reissued_at":"2026-07-05T08:42:01.304850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:42:01.304850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Relying on the Unreliable: The Impact of Language Models' Reluctance to Express Uncertainty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Jena D. Hwang, Kaitlyn Zhou, Maarten Sap, Xiang Ren","submitted_at":"2024-01-12T18:03:30Z","abstract_excerpt":"As natural language becomes the default interface for human-AI interaction, there is a need for LMs to appropriately communicate uncertainties in downstream applications. In this work, we investigate how LMs incorporate confidence in responses via natural language and how downstream users behave in response to LM-articulated uncertainties. We examine publicly deployed models and find that LMs are reluctant to express uncertainties when answering questions even when they produce incorrect responses. LMs can be explicitly prompted to express confidences, but tend to be overconfident, resulting i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06730","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06730/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06730","created_at":"2026-07-05T08:42:01.304909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06730v2","created_at":"2026-07-05T08:42:01.304909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06730","created_at":"2026-07-05T08:42:01.304909+00:00"},{"alias_kind":"pith_short_12","alias_value":"5KJX2UVA7WVK","created_at":"2026-07-05T08:42:01.304909+00:00"},{"alias_kind":"pith_short_16","alias_value":"5KJX2UVA7WVKP2OQ","created_at":"2026-07-05T08:42:01.304909+00:00"},{"alias_kind":"pith_short_8","alias_value":"5KJX2UVA","created_at":"2026-07-05T08:42:01.304909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17312","citing_title":"Quantifying Consistency in LLM Logical Reasoning via Structural Uncertainty","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06673","citing_title":"Uncertainty-Aware LLM-Guided Policy Shaping for Sparse-Reward Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22179","citing_title":"The Score Granularity Gap in Black-Box LLM Classification: A Comparative Study of Confidence Constructions","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23684","citing_title":"Synthetic Sources?: Auditing Generative Search Engine Citations for Evidence of AI-Generated Sources","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08010","citing_title":"Measuring and mitigating overreliance to build human-compatible AI","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2402.05070","citing_title":"A Roadmap to Pluralistic Alignment","ref_index":291,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01302","citing_title":"Beyond Semantic Relevance: Counterfactual Risk Minimization for Robust Retrieval-Augmented Generation","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":105,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K","json":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K.json","graph_json":"https://pith.science/api/pith-number/5KJX2UVA7WVKP2OQ276QQOGY7K/graph.json","events_json":"https://pith.science/api/pith-number/5KJX2UVA7WVKP2OQ276QQOGY7K/events.json","paper":"https://pith.science/paper/5KJX2UVA"},"agent_actions":{"view_html":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K","download_json":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K.json","view_paper":"https://pith.science/paper/5KJX2UVA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06730&json=true","fetch_graph":"https://pith.science/api/pith-number/5KJX2UVA7WVKP2OQ276QQOGY7K/graph.json","fetch_events":"https://pith.science/api/pith-number/5KJX2UVA7WVKP2OQ276QQOGY7K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K/action/storage_attestation","attest_author":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K/action/author_attestation","sign_citation":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K/action/citation_signature","submit_replication":"https://pith.science/pith/5KJX2UVA7WVKP2OQ276QQOGY7K/action/replication_record"}},"created_at":"2026-07-05T08:42:01.304909+00:00","updated_at":"2026-07-05T08:42:01.304909+00:00"}