{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5APEOBXOIM6SLX375T4CI53TDT","short_pith_number":"pith:5APEOBXO","schema_version":"1.0","canonical_sha256":"e81e4706ee433d25df7fecf82477731cd3fca0c74f5a444926f3833f435fdec6","source":{"kind":"arxiv","id":"2412.15296","version":1},"attestation_state":"computed","paper":{"title":"Confidence in the Reasoning of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chris Holmes, Yudi Pawitan","submitted_at":"2024-12-19T10:04:29Z","abstract_excerpt":"There is a growing literature on reasoning by large language models (LLMs), but the discussion on the uncertainty in their responses is still lacking. Our aim is to assess the extent of confidence that LLMs have in their answers and how it correlates with accuracy. Confidence is measured (i) qualitatively in terms of persistence in keeping their answer when prompted to reconsider, and (ii) quantitatively in terms of self-reported confidence score. We investigate the performance of three LLMs -- GPT4o, GPT4-turbo and Mistral -- on two benchmark sets of questions on causal judgement and formal f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15296","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-19T10:04:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e11494b003022ccc2ea50f4e0d17dc7107ca784444dd62cef0ca7dc11ae9fb59","abstract_canon_sha256":"b475ecf7dc98a60cd2da9fee2e14f948258c8eed907c919b25b2bb234da3193e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:20.849384Z","signature_b64":"71M9ZVzoMR4hqK7D/rsgQ7N3sS7zbUtz5uBy2j4A4f4n9hQpRZd485idoL9dMo3GUqs9AUOsu5eznVzv4tBtAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e81e4706ee433d25df7fecf82477731cd3fca0c74f5a444926f3833f435fdec6","last_reissued_at":"2026-07-05T09:52:20.848863Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:20.848863Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Confidence in the Reasoning of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Chris Holmes, Yudi Pawitan","submitted_at":"2024-12-19T10:04:29Z","abstract_excerpt":"There is a growing literature on reasoning by large language models (LLMs), but the discussion on the uncertainty in their responses is still lacking. Our aim is to assess the extent of confidence that LLMs have in their answers and how it correlates with accuracy. Confidence is measured (i) qualitatively in terms of persistence in keeping their answer when prompted to reconsider, and (ii) quantitatively in terms of self-reported confidence score. We investigate the performance of three LLMs -- GPT4o, GPT4-turbo and Mistral -- on two benchmark sets of questions on causal judgement and formal f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15296","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15296/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15296","created_at":"2026-07-05T09:52:20.848926+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15296v1","created_at":"2026-07-05T09:52:20.848926+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15296","created_at":"2026-07-05T09:52:20.848926+00:00"},{"alias_kind":"pith_short_12","alias_value":"5APEOBXOIM6S","created_at":"2026-07-05T09:52:20.848926+00:00"},{"alias_kind":"pith_short_16","alias_value":"5APEOBXOIM6SLX37","created_at":"2026-07-05T09:52:20.848926+00:00"},{"alias_kind":"pith_short_8","alias_value":"5APEOBXO","created_at":"2026-07-05T09:52:20.848926+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.09775","citing_title":"Multiple Choice Questions: Reasoning Makes Large Language Models (LLMs) More Self-Confident, Especially When They are Wrong","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT","json":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT.json","graph_json":"https://pith.science/api/pith-number/5APEOBXOIM6SLX375T4CI53TDT/graph.json","events_json":"https://pith.science/api/pith-number/5APEOBXOIM6SLX375T4CI53TDT/events.json","paper":"https://pith.science/paper/5APEOBXO"},"agent_actions":{"view_html":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT","download_json":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT.json","view_paper":"https://pith.science/paper/5APEOBXO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15296&json=true","fetch_graph":"https://pith.science/api/pith-number/5APEOBXOIM6SLX375T4CI53TDT/graph.json","fetch_events":"https://pith.science/api/pith-number/5APEOBXOIM6SLX375T4CI53TDT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT/action/storage_attestation","attest_author":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT/action/author_attestation","sign_citation":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT/action/citation_signature","submit_replication":"https://pith.science/pith/5APEOBXOIM6SLX375T4CI53TDT/action/replication_record"}},"created_at":"2026-07-05T09:52:20.848926+00:00","updated_at":"2026-07-05T09:52:20.848926+00:00"}