{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:33GVP4ACWZFLG23NHW4XWOHGA3","short_pith_number":"pith:33GVP4AC","schema_version":"1.0","canonical_sha256":"decd57f002b64ab36b6d3db97b38e606d02f8a2f262a768be870796b1f45da90","source":{"kind":"arxiv","id":"2311.09731","version":2},"attestation_state":"computed","paper":{"title":"Examining LLMs' Uncertainty Expression Towards Questions Outside Parametric Knowledge","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Genglin Liu, Hao Peng, Lifan Yuan, Xingyao Wang, Yangyi Chen","submitted_at":"2023-11-16T10:02:40Z","abstract_excerpt":"Can large language models (LLMs) express their uncertainty in situations where they lack sufficient parametric knowledge to generate reasonable responses? This work aims to systematically investigate LLMs' behaviors in such situations, emphasizing the trade-off between honesty and helpfulness. To tackle the challenge of precisely determining LLMs' knowledge gaps, we diagnostically create unanswerable questions containing non-existent concepts or false premises, ensuring that they are outside the LLMs' vast training data. By compiling a benchmark, UnknownBench, which consists of both unanswerab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09731","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-16T10:02:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8069de9f543e66a5d000a3fc8a8ade1dd3dea5c88372c43846cc207414c8cdc1","abstract_canon_sha256":"8d9ef5124bf3413c6bf77fa941357787cc106c31a335c7972358e3f47ee90a2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:45:50.600043Z","signature_b64":"hW/Zt3H7vmq5VHD96BnSEugsxHPbjlSKB61O/2YhuOtiF2tM4iF3Fy9lCik6sHw8amlkXwW41kvfl6bUYbnFAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"decd57f002b64ab36b6d3db97b38e606d02f8a2f262a768be870796b1f45da90","last_reissued_at":"2026-07-05T07:45:50.599341Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:45:50.599341Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Examining LLMs' Uncertainty Expression Towards Questions Outside Parametric Knowledge","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Genglin Liu, Hao Peng, Lifan Yuan, Xingyao Wang, Yangyi Chen","submitted_at":"2023-11-16T10:02:40Z","abstract_excerpt":"Can large language models (LLMs) express their uncertainty in situations where they lack sufficient parametric knowledge to generate reasonable responses? This work aims to systematically investigate LLMs' behaviors in such situations, emphasizing the trade-off between honesty and helpfulness. To tackle the challenge of precisely determining LLMs' knowledge gaps, we diagnostically create unanswerable questions containing non-existent concepts or false premises, ensuring that they are outside the LLMs' vast training data. By compiling a benchmark, UnknownBench, which consists of both unanswerab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09731","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09731/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09731","created_at":"2026-07-05T07:45:50.599418+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09731v2","created_at":"2026-07-05T07:45:50.599418+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09731","created_at":"2026-07-05T07:45:50.599418+00:00"},{"alias_kind":"pith_short_12","alias_value":"33GVP4ACWZFL","created_at":"2026-07-05T07:45:50.599418+00:00"},{"alias_kind":"pith_short_16","alias_value":"33GVP4ACWZFLG23N","created_at":"2026-07-05T07:45:50.599418+00:00"},{"alias_kind":"pith_short_8","alias_value":"33GVP4AC","created_at":"2026-07-05T07:45:50.599418+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19950","citing_title":"Confidence Calibration for Multimodal LLMs: An Empirical Study through Medical VQA","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17324","citing_title":"ASPI: Seeking Ambiguity Clarification Amplifies Prompt Injection Vulnerability in LLM Agents","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10718","citing_title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17843","citing_title":"Learning from AVA: Early Lessons from a Curated and Trustworthy Generative AI for Policy and Development Research","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3","json":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3.json","graph_json":"https://pith.science/api/pith-number/33GVP4ACWZFLG23NHW4XWOHGA3/graph.json","events_json":"https://pith.science/api/pith-number/33GVP4ACWZFLG23NHW4XWOHGA3/events.json","paper":"https://pith.science/paper/33GVP4AC"},"agent_actions":{"view_html":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3","download_json":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3.json","view_paper":"https://pith.science/paper/33GVP4AC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09731&json=true","fetch_graph":"https://pith.science/api/pith-number/33GVP4ACWZFLG23NHW4XWOHGA3/graph.json","fetch_events":"https://pith.science/api/pith-number/33GVP4ACWZFLG23NHW4XWOHGA3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3/action/storage_attestation","attest_author":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3/action/author_attestation","sign_citation":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3/action/citation_signature","submit_replication":"https://pith.science/pith/33GVP4ACWZFLG23NHW4XWOHGA3/action/replication_record"}},"created_at":"2026-07-05T07:45:50.599418+00:00","updated_at":"2026-07-05T07:45:50.599418+00:00"}