{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:F34ZUHHXJVOH2GKD63TUS2HOYZ","short_pith_number":"pith:F34ZUHHX","schema_version":"1.0","canonical_sha256":"2ef99a1cf74d5c7d1943f6e74968eec66952767fe859552bc5d495484dc3ee31","source":{"kind":"arxiv","id":"2305.14975","version":2},"attestation_state":"computed","paper":{"title":"Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Allan Zhou, Archit Sharma, Chelsea Finn, Christopher D. Manning, Eric Mitchell, Huaxiu Yao, Katherine Tian, Rafael Rafailov","submitted_at":"2023-05-24T10:12:33Z","abstract_excerpt":"A trustworthy real-world prediction system should produce well-calibrated confidence scores; that is, its confidence in an answer should be indicative of the likelihood that the answer is correct, enabling deferral to an expert in cases of low-confidence predictions. Recent studies have shown that unsupervised pre-training produces large language models (LMs) whose conditional probabilities are remarkably well-calibrated. However, the most widely-used LMs are fine-tuned with reinforcement learning from human feedback (RLHF-LMs), and some studies have suggested that RLHF-LMs produce conditional"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14975","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T10:12:33Z","cross_cats_sorted":[],"title_canon_sha256":"837ecfd764039a12c3fb792ce31f85dfd77a317a93996b676124e3d864ec5f50","abstract_canon_sha256":"3260444322d49904fe3ebbe0c60508cadb602dc013bf17a21ef23fb273dcb672"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:04:16.316661Z","signature_b64":"bRUNm4srapMLS30u50NW+BjpRiXSlPDzXkF5mRdHTxltuP615q2NE8NtN6UiqXZCysZ+q35BtStSbI6YLn1uBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ef99a1cf74d5c7d1943f6e74968eec66952767fe859552bc5d495484dc3ee31","last_reissued_at":"2026-07-05T07:04:16.316125Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:04:16.316125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Allan Zhou, Archit Sharma, Chelsea Finn, Christopher D. Manning, Eric Mitchell, Huaxiu Yao, Katherine Tian, Rafael Rafailov","submitted_at":"2023-05-24T10:12:33Z","abstract_excerpt":"A trustworthy real-world prediction system should produce well-calibrated confidence scores; that is, its confidence in an answer should be indicative of the likelihood that the answer is correct, enabling deferral to an expert in cases of low-confidence predictions. Recent studies have shown that unsupervised pre-training produces large language models (LMs) whose conditional probabilities are remarkably well-calibrated. However, the most widely-used LMs are fine-tuned with reinforcement learning from human feedback (RLHF-LMs), and some studies have suggested that RLHF-LMs produce conditional"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14975","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14975/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14975","created_at":"2026-07-05T07:04:16.316187+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14975v2","created_at":"2026-07-05T07:04:16.316187+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14975","created_at":"2026-07-05T07:04:16.316187+00:00"},{"alias_kind":"pith_short_12","alias_value":"F34ZUHHXJVOH","created_at":"2026-07-05T07:04:16.316187+00:00"},{"alias_kind":"pith_short_16","alias_value":"F34ZUHHXJVOH2GKD","created_at":"2026-07-05T07:04:16.316187+00:00"},{"alias_kind":"pith_short_8","alias_value":"F34ZUHHX","created_at":"2026-07-05T07:04:16.316187+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06327","citing_title":"Estimating Uncertainty from Reasoning: A Large-Scale Study of Multi- and Crosslingual MCQA Performance in LLMs","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24420","citing_title":"Beyond Logprobs: A Multi-Signal Confidence Engine for LLM-Based Document Field Extraction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27023","citing_title":"Just how sure are you? Improving Verbalized Uncertainty Calibration in Medical VQA","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19950","citing_title":"Confidence Calibration for Multimodal LLMs: An Empirical Study through Medical VQA","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18158","citing_title":"The Measurement Gap in the Automation of EU Law: Benchmarking Doctrinal Legal Reasoning under the EU AI Act","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01612","citing_title":"Scaling with Confidence: Calibrating Confidence of LLMs for Adaptive Test Time Scaling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03535","citing_title":"Can LLM Rerankers Predict Their Own Ranking Performance?","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30826","citing_title":"Beyond Agreement: Scoring Panel-Surfaced Biomedical Entity Candidates for Curator Triage","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22949","citing_title":"MARGIN: Runtime Confidence Calibration for Multi-Agent Foundation Model Coordination","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29490","citing_title":"Reported Confidence in LLMs Tracks Commitment More Than Correctness","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29605","citing_title":"VLAConf: Calibrated Task-Success Confidence for Vision-Language-Action Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28500","citing_title":"Functional Entropy: Predicting Functional Correctness in LLM-Generated Code with Uncertainty Quantification","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00809","citing_title":"NBQ: Next-Best-Question for Dynamic Profiling","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22949","citing_title":"MARGIN: Runtime Confidence Calibration for Multi-Agent Foundation Model Coordination","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06431","citing_title":"Functional-level Uncertainty Quantification for Calibrated Fine-tuning on LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17839","citing_title":"How do LLMs Compute Verbal Confidence","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22161","citing_title":"Causal Evidence that Language Models use Confidence to Drive Behavior","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15416","citing_title":"Margin-Adaptive Confidence Ranking for Reliable LLM Judgement","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18880","citing_title":"Can LLMs Estimate Student Struggles? Human-AI Difficulty Alignment with Proficiency Simulation for Item Difficulty Prediction","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00200","citing_title":"Confidence Estimation in Automatic Short Answer Grading with LLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13595","citing_title":"Inducing Artificial Uncertainty in Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08588","citing_title":"Act or Escalate? Evaluating Escalation Behavior in Automation with Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22215","citing_title":"Verbal Confidence Saturation in 3-9B Open-Weight Instruction-Tuned LLMs: A Pre-Registered Psychometric Validity Screen","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22271","citing_title":"How LLMs Detect and Correct Their Own Errors: The Role of Internal Confidence Signals","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02241","citing_title":"Zero-Shot Confidence Estimation for Small LLMs: When Supervised Baselines Aren't Worth Training","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ","json":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ.json","graph_json":"https://pith.science/api/pith-number/F34ZUHHXJVOH2GKD63TUS2HOYZ/graph.json","events_json":"https://pith.science/api/pith-number/F34ZUHHXJVOH2GKD63TUS2HOYZ/events.json","paper":"https://pith.science/paper/F34ZUHHX"},"agent_actions":{"view_html":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ","download_json":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ.json","view_paper":"https://pith.science/paper/F34ZUHHX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14975&json=true","fetch_graph":"https://pith.science/api/pith-number/F34ZUHHXJVOH2GKD63TUS2HOYZ/graph.json","fetch_events":"https://pith.science/api/pith-number/F34ZUHHXJVOH2GKD63TUS2HOYZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ/action/storage_attestation","attest_author":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ/action/author_attestation","sign_citation":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ/action/citation_signature","submit_replication":"https://pith.science/pith/F34ZUHHXJVOH2GKD63TUS2HOYZ/action/replication_record"}},"created_at":"2026-07-05T07:04:16.316187+00:00","updated_at":"2026-07-05T07:04:16.316187+00:00"}