{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CM6U2UR3FGDDSHFWFDDE7FR5FV","short_pith_number":"pith:CM6U2UR3","schema_version":"1.0","canonical_sha256":"133d4d523b2986391cb628c64f963d2d402feb6dbccbc342ef6ee38384a56a9b","source":{"kind":"arxiv","id":"2505.23854","version":1},"attestation_state":"computed","paper":{"title":"Revisiting Uncertainty Estimation and Calibration of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chang Xu, Linwei Tao, Minjing Dong, Philip Torr, Tao Huang, Yi-Fan Yeh","submitted_at":"2025-05-29T02:04:49Z","abstract_excerpt":"As large language models (LLMs) are increasingly deployed in high-stakes applications, robust uncertainty estimation is essential for ensuring the safe and trustworthy deployment of LLMs. We present the most comprehensive study to date of uncertainty estimation in LLMs, evaluating 80 models spanning open- and closed-source families, dense and Mixture-of-Experts (MoE) architectures, reasoning and non-reasoning modes, quantization variants and parameter scales from 0.6B to 671B. Focusing on three representative black-box single-pass methods, including token probability-based uncertainty (TPU), n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23854","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-29T02:04:49Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"22e72412526b29032dd4cccc5aa19cb5294f860803678ab175acdc5f733101ce","abstract_canon_sha256":"b5476024c125091a037cb67640cbf68b8e106d100492e0c238d8f342653f88e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:36.798048Z","signature_b64":"U8xaNflbn94GYjoWITunA6kPGAL1e230GrzQA4NcOUKm2jdZXXHtGmzzkvE+SNRinrayUa5NMyVSwnXdUYxzBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"133d4d523b2986391cb628c64f963d2d402feb6dbccbc342ef6ee38384a56a9b","last_reissued_at":"2026-07-05T11:12:36.797464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:36.797464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting Uncertainty Estimation and Calibration of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chang Xu, Linwei Tao, Minjing Dong, Philip Torr, Tao Huang, Yi-Fan Yeh","submitted_at":"2025-05-29T02:04:49Z","abstract_excerpt":"As large language models (LLMs) are increasingly deployed in high-stakes applications, robust uncertainty estimation is essential for ensuring the safe and trustworthy deployment of LLMs. We present the most comprehensive study to date of uncertainty estimation in LLMs, evaluating 80 models spanning open- and closed-source families, dense and Mixture-of-Experts (MoE) architectures, reasoning and non-reasoning modes, quantization variants and parameter scales from 0.6B to 671B. Focusing on three representative black-box single-pass methods, including token probability-based uncertainty (TPU), n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23854","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23854/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23854","created_at":"2026-07-05T11:12:36.797535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23854v1","created_at":"2026-07-05T11:12:36.797535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23854","created_at":"2026-07-05T11:12:36.797535+00:00"},{"alias_kind":"pith_short_12","alias_value":"CM6U2UR3FGDD","created_at":"2026-07-05T11:12:36.797535+00:00"},{"alias_kind":"pith_short_16","alias_value":"CM6U2UR3FGDDSHFW","created_at":"2026-07-05T11:12:36.797535+00:00"},{"alias_kind":"pith_short_8","alias_value":"CM6U2UR3","created_at":"2026-07-05T11:12:36.797535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09635","citing_title":"Gradient-Guided Reward Optimization for Inference-time Alignment","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19344","citing_title":"Retrieval-Augmented Linguistic Calibration","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22161","citing_title":"Causal Evidence that Language Models use Confidence to Drive Behavior","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19344","citing_title":"Retrieval-Augmented Linguistic Calibration","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19781","citing_title":"Do Small Language Models Know When They're Wrong? Confidence-Based Cascade Scoring for Educational Assessment","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11328","citing_title":"Epistemic Uncertainty for Test-Time Discovery","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01428","citing_title":"Hallucinations Undermine Trust; Metacognition is a Way Forward","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00957","citing_title":"\"I Don't Know\" -- Towards Appropriate Trust with Certainty-Aware Retrieval Augmented Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05851","citing_title":"Hypothesis generation and updating in large language models","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV","json":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV.json","graph_json":"https://pith.science/api/pith-number/CM6U2UR3FGDDSHFWFDDE7FR5FV/graph.json","events_json":"https://pith.science/api/pith-number/CM6U2UR3FGDDSHFWFDDE7FR5FV/events.json","paper":"https://pith.science/paper/CM6U2UR3"},"agent_actions":{"view_html":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV","download_json":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV.json","view_paper":"https://pith.science/paper/CM6U2UR3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23854&json=true","fetch_graph":"https://pith.science/api/pith-number/CM6U2UR3FGDDSHFWFDDE7FR5FV/graph.json","fetch_events":"https://pith.science/api/pith-number/CM6U2UR3FGDDSHFWFDDE7FR5FV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV/action/storage_attestation","attest_author":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV/action/author_attestation","sign_citation":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV/action/citation_signature","submit_replication":"https://pith.science/pith/CM6U2UR3FGDDSHFWFDDE7FR5FV/action/replication_record"}},"created_at":"2026-07-05T11:12:36.797535+00:00","updated_at":"2026-07-05T11:12:36.797535+00:00"}