{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:SAHVXPK2SPS5DUUMVGPRZE6AXG","short_pith_number":"pith:SAHVXPK2","schema_version":"1.0","canonical_sha256":"900f5bbd5a93e5d1d28ca99f1c93c0b9ad605a6c14bb52a8d15ae2456bfe8916","source":{"kind":"arxiv","id":"2607.19367","version":1},"attestation_state":"computed","paper":{"title":"Rethinking Uncertainty Evaluation in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andy Zou, Atharv Naphade, Krish Matta","submitted_at":"2026-06-11T02:27:49Z","abstract_excerpt":"Calibration is the primary criterion for evaluating LLM confidence, but it is insufficient: it admits trivially incoherent estimators, depends on the evaluation distribution, and does not test the extent to which the estimation can be interpreted as a consistent, underlying probability function. What we actually need is for LLM confidence estimates to satisfy the conditions required of coherent probabilistic beliefs. We formalize these conditions along three axes (structural coherence, faithfulness, and usefulness) and operationalize them as the C1 metrics. Widely used estimators systematicall"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.19367","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-06-11T02:27:49Z","cross_cats_sorted":[],"title_canon_sha256":"7e91caf5616d59a520d84e8ed09ab6573f4685756e62d5ed5807cd7bf0f3ef7a","abstract_canon_sha256":"5991099e7924091519bdd7421efebc036e8fac98da5699079aef9b1e4f8ce39f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-23T00:23:45.130092Z","signature_b64":"y/HK2Fo14yuI3yi5MInuImboxdpPR91lgS+oxmGmyRyNuWCvm37jmiPayYk1JY7SGoLzJ6MY7lwl+5/H/LXXDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"900f5bbd5a93e5d1d28ca99f1c93c0b9ad605a6c14bb52a8d15ae2456bfe8916","last_reissued_at":"2026-07-23T00:23:45.129140Z","signature_status":"signed_v1","first_computed_at":"2026-07-23T00:23:45.129140Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Uncertainty Evaluation in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andy Zou, Atharv Naphade, Krish Matta","submitted_at":"2026-06-11T02:27:49Z","abstract_excerpt":"Calibration is the primary criterion for evaluating LLM confidence, but it is insufficient: it admits trivially incoherent estimators, depends on the evaluation distribution, and does not test the extent to which the estimation can be interpreted as a consistent, underlying probability function. What we actually need is for LLM confidence estimates to satisfy the conditions required of coherent probabilistic beliefs. We formalize these conditions along three axes (structural coherence, faithfulness, and usefulness) and operationalize them as the C1 metrics. Widely used estimators systematicall"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.19367","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.19367/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.19367","created_at":"2026-07-23T00:23:45.129614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.19367v1","created_at":"2026-07-23T00:23:45.129614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.19367","created_at":"2026-07-23T00:23:45.129614+00:00"},{"alias_kind":"pith_short_12","alias_value":"SAHVXPK2SPS5","created_at":"2026-07-23T00:23:45.129614+00:00"},{"alias_kind":"pith_short_16","alias_value":"SAHVXPK2SPS5DUUM","created_at":"2026-07-23T00:23:45.129614+00:00"},{"alias_kind":"pith_short_8","alias_value":"SAHVXPK2","created_at":"2026-07-23T00:23:45.129614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG","json":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG.json","graph_json":"https://pith.science/api/pith-number/SAHVXPK2SPS5DUUMVGPRZE6AXG/graph.json","events_json":"https://pith.science/api/pith-number/SAHVXPK2SPS5DUUMVGPRZE6AXG/events.json","paper":"https://pith.science/paper/SAHVXPK2"},"agent_actions":{"view_html":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG","download_json":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG.json","view_paper":"https://pith.science/paper/SAHVXPK2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.19367&json=true","fetch_graph":"https://pith.science/api/pith-number/SAHVXPK2SPS5DUUMVGPRZE6AXG/graph.json","fetch_events":"https://pith.science/api/pith-number/SAHVXPK2SPS5DUUMVGPRZE6AXG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG/action/storage_attestation","attest_author":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG/action/author_attestation","sign_citation":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG/action/citation_signature","submit_replication":"https://pith.science/pith/SAHVXPK2SPS5DUUMVGPRZE6AXG/action/replication_record"}},"created_at":"2026-07-23T00:23:45.129614+00:00","updated_at":"2026-07-23T00:23:45.129614+00:00"}