{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4HWBXJ2Q2A6DLL7OL4FHODXRXT","short_pith_number":"pith:4HWBXJ2Q","schema_version":"1.0","canonical_sha256":"e1ec1ba750d03c35afee5f0a770ef1bcf3a78b56c246193c30708a0951b4e932","source":{"kind":"arxiv","id":"2311.03533","version":1},"attestation_state":"computed","paper":{"title":"Quantifying Uncertainty in Natural Language Explanations of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chirag Agarwal, Himabindu Lakkaraju, Sree Harsha Tanneru","submitted_at":"2023-11-06T21:14:40Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used as powerful tools for several high-stakes natural language processing (NLP) applications. Recent prompting works claim to elicit intermediate reasoning steps and key tokens that serve as proxy explanations for LLM predictions. However, there is no certainty whether these explanations are reliable and reflect the LLMs behavior. In this work, we make one of the first attempts at quantifying the uncertainty in explanations of LLMs. To this end, we propose two novel metrics -- $\\textit{Verbalized Uncertainty}$ and $\\textit{Probing Uncertainty}$ --"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.03533","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-06T21:14:40Z","cross_cats_sorted":[],"title_canon_sha256":"49e50d297ac680beba66fc50f755789cf17409e5ace361d3cc14faac64ce33c5","abstract_canon_sha256":"fd183c1db094f111573892b243458cc8b94e8efc72479e20919c7a360732aac8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:59.759499Z","signature_b64":"t4E1bthd2+4lSYh6mebgV+Xm4LEZRTAB/PVp8vx4anq/65+i8bMffQVoxzENbQhTiXcLMJQhCrKYka9IW4jHDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1ec1ba750d03c35afee5f0a770ef1bcf3a78b56c246193c30708a0951b4e932","last_reissued_at":"2026-07-05T07:09:59.758853Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:59.758853Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying Uncertainty in Natural Language Explanations of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chirag Agarwal, Himabindu Lakkaraju, Sree Harsha Tanneru","submitted_at":"2023-11-06T21:14:40Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used as powerful tools for several high-stakes natural language processing (NLP) applications. Recent prompting works claim to elicit intermediate reasoning steps and key tokens that serve as proxy explanations for LLM predictions. However, there is no certainty whether these explanations are reliable and reflect the LLMs behavior. In this work, we make one of the first attempts at quantifying the uncertainty in explanations of LLMs. To this end, we propose two novel metrics -- $\\textit{Verbalized Uncertainty}$ and $\\textit{Probing Uncertainty}$ --"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.03533","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.03533/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.03533","created_at":"2026-07-05T07:09:59.758968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.03533v1","created_at":"2026-07-05T07:09:59.758968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.03533","created_at":"2026-07-05T07:09:59.758968+00:00"},{"alias_kind":"pith_short_12","alias_value":"4HWBXJ2Q2A6D","created_at":"2026-07-05T07:09:59.758968+00:00"},{"alias_kind":"pith_short_16","alias_value":"4HWBXJ2Q2A6DLL7O","created_at":"2026-07-05T07:09:59.758968+00:00"},{"alias_kind":"pith_short_8","alias_value":"4HWBXJ2Q","created_at":"2026-07-05T07:09:59.758968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.15906","citing_title":"Towards Reliable, Uncertainty-Aware Alignment","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT","json":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT.json","graph_json":"https://pith.science/api/pith-number/4HWBXJ2Q2A6DLL7OL4FHODXRXT/graph.json","events_json":"https://pith.science/api/pith-number/4HWBXJ2Q2A6DLL7OL4FHODXRXT/events.json","paper":"https://pith.science/paper/4HWBXJ2Q"},"agent_actions":{"view_html":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT","download_json":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT.json","view_paper":"https://pith.science/paper/4HWBXJ2Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.03533&json=true","fetch_graph":"https://pith.science/api/pith-number/4HWBXJ2Q2A6DLL7OL4FHODXRXT/graph.json","fetch_events":"https://pith.science/api/pith-number/4HWBXJ2Q2A6DLL7OL4FHODXRXT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT/action/storage_attestation","attest_author":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT/action/author_attestation","sign_citation":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT/action/citation_signature","submit_replication":"https://pith.science/pith/4HWBXJ2Q2A6DLL7OL4FHODXRXT/action/replication_record"}},"created_at":"2026-07-05T07:09:59.758968+00:00","updated_at":"2026-07-05T07:09:59.758968+00:00"}