{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A6QJCWHRZEOE3VNCS55GKBEJZ7","short_pith_number":"pith:A6QJCWHR","schema_version":"1.0","canonical_sha256":"07a09158f1c91c4dd5a2977a650489cfccd66d382a4a33358c0f6e87c7d36a04","source":{"kind":"arxiv","id":"2411.13343","version":1},"attestation_state":"computed","paper":{"title":"Fact-Level Confidence Calibration and Self-Correction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bingbing Xu, Fei Sun, Hexiang Tan, Huawei Shen, Teng Xiao, Wei Li, Xueqi Cheng, Yige Yuan","submitted_at":"2024-11-20T14:15:18Z","abstract_excerpt":"Confidence calibration in LLMs, i.e., aligning their self-assessed confidence with the actual accuracy of their responses, enabling them to self-evaluate the correctness of their outputs. However, current calibration methods for LLMs typically estimate two scalars to represent overall response confidence and correctness, which is inadequate for long-form generation where the response includes multiple atomic facts and may be partially confident and correct. These methods also overlook the relevance of each fact to the query. To address these challenges, we propose a Fact-Level Calibration fram"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.13343","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-20T14:15:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"169805c6e0bea20010b8dba1b5bdb45812df9819b498ff05499b83de72ee0dc4","abstract_canon_sha256":"0e15b05f1993f0e197126958fe204cfed7f2934d3715b147e4413f3fa13b112c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:09.078099Z","signature_b64":"f/Cq5WUD+ZzIZhXUDI993zMzuT639z9q50rRmNQSp9NkkmtvOA+A8QHQip5vUmtp9Is/FeFUsd2DKnPnYy+dBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07a09158f1c91c4dd5a2977a650489cfccd66d382a4a33358c0f6e87c7d36a04","last_reissued_at":"2026-07-05T09:38:09.077624Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:09.077624Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fact-Level Confidence Calibration and Self-Correction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bingbing Xu, Fei Sun, Hexiang Tan, Huawei Shen, Teng Xiao, Wei Li, Xueqi Cheng, Yige Yuan","submitted_at":"2024-11-20T14:15:18Z","abstract_excerpt":"Confidence calibration in LLMs, i.e., aligning their self-assessed confidence with the actual accuracy of their responses, enabling them to self-evaluate the correctness of their outputs. However, current calibration methods for LLMs typically estimate two scalars to represent overall response confidence and correctness, which is inadequate for long-form generation where the response includes multiple atomic facts and may be partially confident and correct. These methods also overlook the relevance of each fact to the query. To address these challenges, we propose a Fact-Level Calibration fram"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.13343","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.13343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.13343","created_at":"2026-07-05T09:38:09.077680+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.13343v1","created_at":"2026-07-05T09:38:09.077680+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.13343","created_at":"2026-07-05T09:38:09.077680+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6QJCWHRZEOE","created_at":"2026-07-05T09:38:09.077680+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6QJCWHRZEOE3VNC","created_at":"2026-07-05T09:38:09.077680+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6QJCWHR","created_at":"2026-07-05T09:38:09.077680+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14186","citing_title":"LLMs Know When They Know, but Do Not Act on It: A Metacognitive Harness for Test-time Scaling","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7","json":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7.json","graph_json":"https://pith.science/api/pith-number/A6QJCWHRZEOE3VNCS55GKBEJZ7/graph.json","events_json":"https://pith.science/api/pith-number/A6QJCWHRZEOE3VNCS55GKBEJZ7/events.json","paper":"https://pith.science/paper/A6QJCWHR"},"agent_actions":{"view_html":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7","download_json":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7.json","view_paper":"https://pith.science/paper/A6QJCWHR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.13343&json=true","fetch_graph":"https://pith.science/api/pith-number/A6QJCWHRZEOE3VNCS55GKBEJZ7/graph.json","fetch_events":"https://pith.science/api/pith-number/A6QJCWHRZEOE3VNCS55GKBEJZ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7/action/storage_attestation","attest_author":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7/action/author_attestation","sign_citation":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7/action/citation_signature","submit_replication":"https://pith.science/pith/A6QJCWHRZEOE3VNCS55GKBEJZ7/action/replication_record"}},"created_at":"2026-07-05T09:38:09.077680+00:00","updated_at":"2026-07-05T09:38:09.077680+00:00"}