{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:W5VR4VBI2HR34EPHIZJIOB7HN5","short_pith_number":"pith:W5VR4VBI","schema_version":"1.0","canonical_sha256":"b76b1e5428d1e3be11e746528707e76f5b934e501991735be20089de8d50a07b","source":{"kind":"arxiv","id":"2305.19187","version":3},"attestation_state":"computed","paper":{"title":"Generating with Confidence: Uncertainty Quantification for Black-box Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Jimeng Sun, Shubhendu Trivedi, Zhen Lin","submitted_at":"2023-05-30T16:31:26Z","abstract_excerpt":"Large language models (LLMs) specializing in natural language generation (NLG) have recently started exhibiting promising capabilities across a variety of domains. However, gauging the trustworthiness of responses generated by LLMs remains an open challenge, with limited research on uncertainty quantification (UQ) for NLG. Furthermore, existing literature typically assumes white-box access to language models, which is becoming unrealistic either due to the closed-source nature of the latest LLMs or computational constraints. In this work, we investigate UQ in NLG for *black-box* LLMs. We first"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.19187","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-30T16:31:26Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"f98286f4f2c63ea75a183bd92178a1ced7591baf888752360fa0d8d40548fa52","abstract_canon_sha256":"82511187ef0c0435d076f5ccbbbe4ec55efe74dc8fdc6f9cc357432e0a197175"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:22.926532Z","signature_b64":"khwnV+blgB4NsXTW3dUT5zBhhG1V4X2ohRj4SMrXopMbcKa3kM9CI9e1PFS2g0fwPw5eG2+dpu7qWvQYmp8MBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b76b1e5428d1e3be11e746528707e76f5b934e501991735be20089de8d50a07b","last_reissued_at":"2026-07-05T08:20:22.926022Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:22.926022Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generating with Confidence: Uncertainty Quantification for Black-box Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Jimeng Sun, Shubhendu Trivedi, Zhen Lin","submitted_at":"2023-05-30T16:31:26Z","abstract_excerpt":"Large language models (LLMs) specializing in natural language generation (NLG) have recently started exhibiting promising capabilities across a variety of domains. However, gauging the trustworthiness of responses generated by LLMs remains an open challenge, with limited research on uncertainty quantification (UQ) for NLG. Furthermore, existing literature typically assumes white-box access to language models, which is becoming unrealistic either due to the closed-source nature of the latest LLMs or computational constraints. In this work, we investigate UQ in NLG for *black-box* LLMs. We first"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.19187","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.19187/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.19187","created_at":"2026-07-05T08:20:22.926082+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.19187v3","created_at":"2026-07-05T08:20:22.926082+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.19187","created_at":"2026-07-05T08:20:22.926082+00:00"},{"alias_kind":"pith_short_12","alias_value":"W5VR4VBI2HR3","created_at":"2026-07-05T08:20:22.926082+00:00"},{"alias_kind":"pith_short_16","alias_value":"W5VR4VBI2HR34EPH","created_at":"2026-07-05T08:20:22.926082+00:00"},{"alias_kind":"pith_short_8","alias_value":"W5VR4VBI","created_at":"2026-07-05T08:20:22.926082+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06327","citing_title":"Estimating Uncertainty from Reasoning: A Large-Scale Study of Multi- and Crosslingual MCQA Performance in LLMs","ref_index":55,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17312","citing_title":"Quantifying Consistency in LLM Logical Reasoning via Structural Uncertainty","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20662","citing_title":"Confidence Laundering in Agent Systems: Why Uncertainty Needs a Latent Carrier","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03087","citing_title":"Learning to Solve, Forgetting to Retain: Correct-Set Turnover in RLVR","ref_index":193,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02093","citing_title":"The Role of Ambiguity in Error Prediction via Uncertainty Quantification","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30814","citing_title":"When Calibration Rankings Reverse: Accuracy-Controlled Evaluation for Fair Comparison of LLMs","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19351","citing_title":"Detecting Hallucinations for Large Language Model-based Knowledge Graph Reasoning","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28500","citing_title":"Functional Entropy: Predicting Functional Correctness in LLM-Generated Code with Uncertainty Quantification","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22864","citing_title":"Reading Calibrated Uncertainty from Language Model Trajectories","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23043","citing_title":"HawkesLLM: Semantic Uncertainty Propagation in Agentic Text Simulation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2408.10692","citing_title":"Unconditional Truthfulness: Learning Unconditional Uncertainty of Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2505.11737","citing_title":"TokUR: Token-Level Uncertainty Estimation for Large Language Model Reasoning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19220","citing_title":"Position: Uncertainty Quantification in LLMs is Just Unsupervised Clustering","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20084","citing_title":"BalanceRAG: Joint Risk Calibration for Cascaded Retrieval-Augmented Generation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18473","citing_title":"Principled Detection of Hallucinations in Large Language Models via Multiple Testing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04351","citing_title":"Distance Is All You Need: Radial Dispersion for Uncertainty Estimation in Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2601.17467","citing_title":"Harnessing Reasoning Trajectories for Hallucination Detection via Answer-agreement Representation Shaping","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09117","citing_title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13277","citing_title":"Utility-Oriented Visual Evidence Selection for Multimodal Retrieval-Augmented Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02543","citing_title":"Overconfidence and Calibration in Medical VQA: Empirical Findings and Hallucination-Aware Mitigation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2309.01219","citing_title":"Siren's Song in the AI Ocean: A Survey on Hallucination in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09023","citing_title":"Using Semantic Distance to Estimate Uncertainty in LLM-Based Code Generation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06308","citing_title":"Measuring Black-Box Confidence via Reasoning Trajectories: Geometry, Coverage, and Verbalization","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05777","citing_title":"Estimating the Black-box LLM Uncertainty with Distribution-Aligned Adversarial Distillation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08708","citing_title":"Every Response Counts: Quantifying Uncertainty of LLM-based Multi-Agent Systems through Tensor Decomposition","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5","json":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5.json","graph_json":"https://pith.science/api/pith-number/W5VR4VBI2HR34EPHIZJIOB7HN5/graph.json","events_json":"https://pith.science/api/pith-number/W5VR4VBI2HR34EPHIZJIOB7HN5/events.json","paper":"https://pith.science/paper/W5VR4VBI"},"agent_actions":{"view_html":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5","download_json":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5.json","view_paper":"https://pith.science/paper/W5VR4VBI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.19187&json=true","fetch_graph":"https://pith.science/api/pith-number/W5VR4VBI2HR34EPHIZJIOB7HN5/graph.json","fetch_events":"https://pith.science/api/pith-number/W5VR4VBI2HR34EPHIZJIOB7HN5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5/action/storage_attestation","attest_author":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5/action/author_attestation","sign_citation":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5/action/citation_signature","submit_replication":"https://pith.science/pith/W5VR4VBI2HR34EPHIZJIOB7HN5/action/replication_record"}},"created_at":"2026-07-05T08:20:22.926082+00:00","updated_at":"2026-07-05T08:20:22.926082+00:00"}