{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BLYUXNVX474BMZ62ON7YJHJBUG","short_pith_number":"pith:BLYUXNVX","schema_version":"1.0","canonical_sha256":"0af14bb6b7e7f81667da737f849d21a1b6367c032ebfb69bb4d929095e1b2208","source":{"kind":"arxiv","id":"2410.09724","version":2},"attestation_state":"computed","paper":{"title":"Taming Overconfidence in LLMs: Reward Calibration in RLHF","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Banghua Zhu, Chengsong Huang, Jiaxin Huang, Jixuan Leng","submitted_at":"2024-10-13T04:48:40Z","abstract_excerpt":"Language model calibration refers to the alignment between the confidence of the model and the actual performance of its responses. While previous studies point out the overconfidence phenomenon in Large Language Models (LLMs) and show that LLMs trained with Reinforcement Learning from Human Feedback (RLHF) are overconfident with a more sharpened output probability, in this study, we reveal that RLHF tends to lead models to express verbalized overconfidence in their own responses. We investigate the underlying cause of this overconfidence and demonstrate that reward models used for Proximal Po"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09724","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-13T04:48:40Z","cross_cats_sorted":[],"title_canon_sha256":"351c4120b8f965991962fee16df2551675e83b569042e6856f51f7c3a79efb8f","abstract_canon_sha256":"7698befe89f2f7b271040ae8059f6159a5dc26aa6cc2616b27c86f28658f99e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:07.485938Z","signature_b64":"9REK7VoE5gz7NMIPRXVIF3Z3pguK2+MU/h71OjmkiS2HmgSIWoJCEs2Tvf1eirOk12Qh6A7+mHO+yUtwCFG3DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0af14bb6b7e7f81667da737f849d21a1b6367c032ebfb69bb4d929095e1b2208","last_reissued_at":"2026-07-05T10:22:07.485374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:07.485374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taming Overconfidence in LLMs: Reward Calibration in RLHF","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Banghua Zhu, Chengsong Huang, Jiaxin Huang, Jixuan Leng","submitted_at":"2024-10-13T04:48:40Z","abstract_excerpt":"Language model calibration refers to the alignment between the confidence of the model and the actual performance of its responses. While previous studies point out the overconfidence phenomenon in Large Language Models (LLMs) and show that LLMs trained with Reinforcement Learning from Human Feedback (RLHF) are overconfident with a more sharpened output probability, in this study, we reveal that RLHF tends to lead models to express verbalized overconfidence in their own responses. We investigate the underlying cause of this overconfidence and demonstrate that reward models used for Proximal Po"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09724","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09724/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09724","created_at":"2026-07-05T10:22:07.485440+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09724v2","created_at":"2026-07-05T10:22:07.485440+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09724","created_at":"2026-07-05T10:22:07.485440+00:00"},{"alias_kind":"pith_short_12","alias_value":"BLYUXNVX474B","created_at":"2026-07-05T10:22:07.485440+00:00"},{"alias_kind":"pith_short_16","alias_value":"BLYUXNVX474BMZ62","created_at":"2026-07-05T10:22:07.485440+00:00"},{"alias_kind":"pith_short_8","alias_value":"BLYUXNVX","created_at":"2026-07-05T10:22:07.485440+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01612","citing_title":"Scaling with Confidence: Calibrating Confidence of LLMs for Adaptive Test Time Scaling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00164","citing_title":"Verifiable Rewards for Calibrated Probabilistic Forecasting","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05436","citing_title":"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32032","citing_title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31371","citing_title":"Calibrating the Evaluator: Does Probability Calibration Mitigate Preference Coupling in LLM Agent Feedback Loops?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21882","citing_title":"Position: The Hidden Costs and Measurement Gaps of Reinforcement Learning with Verifiable Rewards","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05929","citing_title":"LLM Harms: A Taxonomy and Discussion","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09117","citing_title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12718","citing_title":"CHAL: Council of Hierarchical Agentic Language","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11436","citing_title":"Agent-BRACE: Decoupling Beliefs from Actions in Long-Horizon Tasks via Verbalized State Uncertainty","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08974","citing_title":"Confident in a Confidence Score: Investigating the Sensitivity of Confidence Scores to Supervised Fine-Tuning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14651","citing_title":"CURA: Clinical Uncertainty Risk Alignment for Language Model-Based Risk Prediction","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15602","citing_title":"GroupDPO: Memory efficient Group-wise Direct Preference Optimization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19444","citing_title":"Unsupervised Confidence Calibration for Reasoning LLMs from a Single Generation","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20500","citing_title":"Efficient Test-Time Inference via Deterministic Exploration of Truncated Decoding Trees","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG","json":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG.json","graph_json":"https://pith.science/api/pith-number/BLYUXNVX474BMZ62ON7YJHJBUG/graph.json","events_json":"https://pith.science/api/pith-number/BLYUXNVX474BMZ62ON7YJHJBUG/events.json","paper":"https://pith.science/paper/BLYUXNVX"},"agent_actions":{"view_html":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG","download_json":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG.json","view_paper":"https://pith.science/paper/BLYUXNVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09724&json=true","fetch_graph":"https://pith.science/api/pith-number/BLYUXNVX474BMZ62ON7YJHJBUG/graph.json","fetch_events":"https://pith.science/api/pith-number/BLYUXNVX474BMZ62ON7YJHJBUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG/action/storage_attestation","attest_author":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG/action/author_attestation","sign_citation":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG/action/citation_signature","submit_replication":"https://pith.science/pith/BLYUXNVX474BMZ62ON7YJHJBUG/action/replication_record"}},"created_at":"2026-07-05T10:22:07.485440+00:00","updated_at":"2026-07-05T10:22:07.485440+00:00"}