{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RW7E2CWBHFO6ITO2G3ZLPSBKFM","short_pith_number":"pith:RW7E2CWB","schema_version":"1.0","canonical_sha256":"8dbe4d0ac1395de44dda36f2b7c82a2b18bc25f9af7e78cd2391c277ca3a59d6","source":{"kind":"arxiv","id":"2410.06707","version":1},"attestation_state":"computed","paper":{"title":"Calibrating Verbalized Probabilities for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cheng Wang, Georges Balazs, Gyuri Szarvas, Patrick Ernst, Pavel Danchenko","submitted_at":"2024-10-09T09:20:24Z","abstract_excerpt":"Calibrating verbalized probabilities presents a novel approach for reliably assessing and leveraging outputs from black-box Large Language Models (LLMs). Recent methods have demonstrated improved calibration by applying techniques like Platt scaling or temperature scaling to the confidence scores generated by LLMs. In this paper, we explore the calibration of verbalized probability distributions for discriminative tasks. First, we investigate the capability of LLMs to generate probability distributions over categorical labels. We theoretically and empirically identify the issue of re-softmax a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.06707","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-09T09:20:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b671fcf2098a589012fbd31d9dcb2bb0ec439f6bde3eba55f157ba28f322c333","abstract_canon_sha256":"c1a2dee63d5bc3cc350d2333a01a6f52a0c50987d23b371d254b4f71033ea0da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:05.861691Z","signature_b64":"F+sXmddmlOd6QqTR8UCocZ5Yi2yJIf6arD3ab20rWfoerGWoXGxNdi0BTX8m20BRoB4LREwDWNoi3h2ivI3DDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8dbe4d0ac1395de44dda36f2b7c82a2b18bc25f9af7e78cd2391c277ca3a59d6","last_reissued_at":"2026-07-05T09:18:05.861187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:05.861187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Calibrating Verbalized Probabilities for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cheng Wang, Georges Balazs, Gyuri Szarvas, Patrick Ernst, Pavel Danchenko","submitted_at":"2024-10-09T09:20:24Z","abstract_excerpt":"Calibrating verbalized probabilities presents a novel approach for reliably assessing and leveraging outputs from black-box Large Language Models (LLMs). Recent methods have demonstrated improved calibration by applying techniques like Platt scaling or temperature scaling to the confidence scores generated by LLMs. In this paper, we explore the calibration of verbalized probability distributions for discriminative tasks. First, we investigate the capability of LLMs to generate probability distributions over categorical labels. We theoretically and empirically identify the issue of re-softmax a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.06707","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.06707/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.06707","created_at":"2026-07-05T09:18:05.861265+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.06707v1","created_at":"2026-07-05T09:18:05.861265+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06707","created_at":"2026-07-05T09:18:05.861265+00:00"},{"alias_kind":"pith_short_12","alias_value":"RW7E2CWBHFO6","created_at":"2026-07-05T09:18:05.861265+00:00"},{"alias_kind":"pith_short_16","alias_value":"RW7E2CWBHFO6ITO2","created_at":"2026-07-05T09:18:05.861265+00:00"},{"alias_kind":"pith_short_8","alias_value":"RW7E2CWB","created_at":"2026-07-05T09:18:05.861265+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19950","citing_title":"Confidence Calibration for Multimodal LLMs: An Empirical Study through Medical VQA","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17234","citing_title":"Speaking in Self-Assessing Tongues: On the Verbalized Confidence of LLMs in Machine Translation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23333","citing_title":"Process Supervision of Confidence Margin for Calibrated LLM Reasoning","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17725","citing_title":"RePrompT: Recurrent Prompt Tuning for Integrating Structured EHR Encoders with Large Language Models","ref_index":103,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM","json":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM.json","graph_json":"https://pith.science/api/pith-number/RW7E2CWBHFO6ITO2G3ZLPSBKFM/graph.json","events_json":"https://pith.science/api/pith-number/RW7E2CWBHFO6ITO2G3ZLPSBKFM/events.json","paper":"https://pith.science/paper/RW7E2CWB"},"agent_actions":{"view_html":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM","download_json":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM.json","view_paper":"https://pith.science/paper/RW7E2CWB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.06707&json=true","fetch_graph":"https://pith.science/api/pith-number/RW7E2CWBHFO6ITO2G3ZLPSBKFM/graph.json","fetch_events":"https://pith.science/api/pith-number/RW7E2CWBHFO6ITO2G3ZLPSBKFM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM/action/storage_attestation","attest_author":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM/action/author_attestation","sign_citation":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM/action/citation_signature","submit_replication":"https://pith.science/pith/RW7E2CWBHFO6ITO2G3ZLPSBKFM/action/replication_record"}},"created_at":"2026-07-05T09:18:05.861265+00:00","updated_at":"2026-07-05T09:18:05.861265+00:00"}