{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZVMLPDNZPWO7KIG2ACUYJDWA3D","short_pith_number":"pith:ZVMLPDNZ","schema_version":"1.0","canonical_sha256":"cd58b78db97d9df520da00a9848ec0d8e2a91fc4f53cde4423821ddccd732d02","source":{"kind":"arxiv","id":"2404.17287","version":3},"attestation_state":"computed","paper":{"title":"When to Trust LLMs: Aligning Confidence with Response Quality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bolin Ding, Fei Sun, Hanxing Ding, Huawei Shen, Jinyang Gao, Liuyi Yao, Qi Cao, Shuchang Tao, Yuexiang Xie","submitted_at":"2024-04-26T09:42:46Z","abstract_excerpt":"Despite the success of large language models (LLMs) in natural language generation, much evidence shows that LLMs may produce incorrect or nonsensical text. This limitation highlights the importance of discerning when to trust LLMs, especially in safety-critical domains. Existing methods often express reliability by confidence level, however, their effectiveness is limited by the lack of objective guidance. To address this, we propose CONfidence-Quality-ORDer-preserving alignment approach (CONQORD), which leverages reinforcement learning guided by a tailored dual-component reward function. Thi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.17287","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-26T09:42:46Z","cross_cats_sorted":[],"title_canon_sha256":"6e4df1c216d46a008800288021e38d359cef40687aeb5f9c35ed0a4af30aa2f4","abstract_canon_sha256":"13dfc66d7911f1e582685f165d1246d30c94fb773caf81142b385090630502ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:02.118097Z","signature_b64":"/qgC+o4Fd1O4sfz0ryFFhr0Fy/QA0EGfa4mWJhD23ugrdsXlxj71LEqb+HXqU2O8E1l16ontiS/4+wgqsAhWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd58b78db97d9df520da00a9848ec0d8e2a91fc4f53cde4423821ddccd732d02","last_reissued_at":"2026-07-05T09:13:02.117662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:02.117662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When to Trust LLMs: Aligning Confidence with Response Quality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bolin Ding, Fei Sun, Hanxing Ding, Huawei Shen, Jinyang Gao, Liuyi Yao, Qi Cao, Shuchang Tao, Yuexiang Xie","submitted_at":"2024-04-26T09:42:46Z","abstract_excerpt":"Despite the success of large language models (LLMs) in natural language generation, much evidence shows that LLMs may produce incorrect or nonsensical text. This limitation highlights the importance of discerning when to trust LLMs, especially in safety-critical domains. Existing methods often express reliability by confidence level, however, their effectiveness is limited by the lack of objective guidance. To address this, we propose CONfidence-Quality-ORDer-preserving alignment approach (CONQORD), which leverages reinforcement learning guided by a tailored dual-component reward function. Thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.17287","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.17287/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.17287","created_at":"2026-07-05T09:13:02.117713+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.17287v3","created_at":"2026-07-05T09:13:02.117713+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.17287","created_at":"2026-07-05T09:13:02.117713+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZVMLPDNZPWO7","created_at":"2026-07-05T09:13:02.117713+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZVMLPDNZPWO7KIG2","created_at":"2026-07-05T09:13:02.117713+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZVMLPDNZ","created_at":"2026-07-05T09:13:02.117713+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19950","citing_title":"Confidence Calibration for Multimodal LLMs: An Empirical Study through Medical VQA","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28571","citing_title":"Not All Uncertainty Is Equal: How Uncertainty Granularity Shapes Human Verification in LLM-Assisted Decision Making","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12632","citing_title":"Calibration-Aware Policy Optimization for Reasoning LLMs","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D","json":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D.json","graph_json":"https://pith.science/api/pith-number/ZVMLPDNZPWO7KIG2ACUYJDWA3D/graph.json","events_json":"https://pith.science/api/pith-number/ZVMLPDNZPWO7KIG2ACUYJDWA3D/events.json","paper":"https://pith.science/paper/ZVMLPDNZ"},"agent_actions":{"view_html":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D","download_json":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D.json","view_paper":"https://pith.science/paper/ZVMLPDNZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.17287&json=true","fetch_graph":"https://pith.science/api/pith-number/ZVMLPDNZPWO7KIG2ACUYJDWA3D/graph.json","fetch_events":"https://pith.science/api/pith-number/ZVMLPDNZPWO7KIG2ACUYJDWA3D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D/action/storage_attestation","attest_author":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D/action/author_attestation","sign_citation":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D/action/citation_signature","submit_replication":"https://pith.science/pith/ZVMLPDNZPWO7KIG2ACUYJDWA3D/action/replication_record"}},"created_at":"2026-07-05T09:13:02.117713+00:00","updated_at":"2026-07-05T09:13:02.117713+00:00"}