{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:64BT7HGN44KRMHU4DWR5EIJD2G","short_pith_number":"pith:64BT7HGN","schema_version":"1.0","canonical_sha256":"f7033f9ccde715161e9c1da3d22123d1a39b0cd8fd78cb4a28573919cb81cff1","source":{"kind":"arxiv","id":"2403.08819","version":2},"attestation_state":"computed","paper":{"title":"Thermometer: Towards Universal Calibration for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Gregory Wornell, Kristjan Greenewald, Maohao Shen, Prasanna Sattigeri, Soumya Ghosh, Subhro Das","submitted_at":"2024-02-20T04:13:48Z","abstract_excerpt":"We consider the issue of calibration in large language models (LLM). Recent studies have found that common interventions such as instruction tuning often result in poorly calibrated LLMs. Although calibration is well-explored in traditional applications, calibrating LLMs is uniquely challenging. These challenges stem as much from the severe computational requirements of LLMs as from their versatility, which allows them to be applied to diverse tasks. Addressing these challenges, we propose THERMOMETER, a calibration approach tailored to LLMs. THERMOMETER learns an auxiliary model, given data f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08819","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-20T04:13:48Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"6befb37658441f40bdc9739ef9ca4bc272526fb8125b95fbe39f8ed9fe9e4c81","abstract_canon_sha256":"0e5600c0522b6202a55e517e655f04d1b1fcd7d9adc7f94dea68c1f938993ff9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:18.667283Z","signature_b64":"h6goERq4lVbJi6Ly8D2UE2OkLZHLhqVUCTLg2hIBVFpEWurd6sE1mNrmqd5Tw26vsl/xvJxq3bjKLTlmvL84DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7033f9ccde715161e9c1da3d22123d1a39b0cd8fd78cb4a28573919cb81cff1","last_reissued_at":"2026-07-05T08:37:18.666786Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:18.666786Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thermometer: Towards Universal Calibration for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Gregory Wornell, Kristjan Greenewald, Maohao Shen, Prasanna Sattigeri, Soumya Ghosh, Subhro Das","submitted_at":"2024-02-20T04:13:48Z","abstract_excerpt":"We consider the issue of calibration in large language models (LLM). Recent studies have found that common interventions such as instruction tuning often result in poorly calibrated LLMs. Although calibration is well-explored in traditional applications, calibrating LLMs is uniquely challenging. These challenges stem as much from the severe computational requirements of LLMs as from their versatility, which allows them to be applied to diverse tasks. Addressing these challenges, we propose THERMOMETER, a calibration approach tailored to LLMs. THERMOMETER learns an auxiliary model, given data f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08819","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08819/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08819","created_at":"2026-07-05T08:37:18.666845+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08819v2","created_at":"2026-07-05T08:37:18.666845+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08819","created_at":"2026-07-05T08:37:18.666845+00:00"},{"alias_kind":"pith_short_12","alias_value":"64BT7HGN44KR","created_at":"2026-07-05T08:37:18.666845+00:00"},{"alias_kind":"pith_short_16","alias_value":"64BT7HGN44KRMHU4","created_at":"2026-07-05T08:37:18.666845+00:00"},{"alias_kind":"pith_short_8","alias_value":"64BT7HGN","created_at":"2026-07-05T08:37:18.666845+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.32032","citing_title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28778","citing_title":"Can LLMs Use Linguistic Uncertainty Markers to Reliably Reflect Intrinsic Confidence?","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13595","citing_title":"Inducing Artificial Uncertainty in Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02543","citing_title":"Overconfidence and Calibration in Medical VQA: Empirical Findings and Hallucination-Aware Mitigation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19895","citing_title":"Learning When Not to Decide: A Framework for Overcoming Factual Presumptuousness in AI Adjudication","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G","json":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G.json","graph_json":"https://pith.science/api/pith-number/64BT7HGN44KRMHU4DWR5EIJD2G/graph.json","events_json":"https://pith.science/api/pith-number/64BT7HGN44KRMHU4DWR5EIJD2G/events.json","paper":"https://pith.science/paper/64BT7HGN"},"agent_actions":{"view_html":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G","download_json":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G.json","view_paper":"https://pith.science/paper/64BT7HGN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08819&json=true","fetch_graph":"https://pith.science/api/pith-number/64BT7HGN44KRMHU4DWR5EIJD2G/graph.json","fetch_events":"https://pith.science/api/pith-number/64BT7HGN44KRMHU4DWR5EIJD2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G/action/storage_attestation","attest_author":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G/action/author_attestation","sign_citation":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G/action/citation_signature","submit_replication":"https://pith.science/pith/64BT7HGN44KRMHU4DWR5EIJD2G/action/replication_record"}},"created_at":"2026-07-05T08:37:18.666845+00:00","updated_at":"2026-07-05T08:37:18.666845+00:00"}