{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:H7IOBD4NK7C5XIYRBNSYJZN5BZ","short_pith_number":"pith:H7IOBD4N","schema_version":"1.0","canonical_sha256":"3fd0e08f8d57c5dba3110b6584e5bd0e42ac56c2a5a95da892b13801c85739e2","source":{"kind":"arxiv","id":"2502.00290","version":5},"attestation_state":"computed","paper":{"title":"Estimating LLM Uncertainty with Evidence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Changqing Zhang, Guangyu Wang, Huan Ma, Jingdong Chen, Joey Tianyi Zhou","submitted_at":"2025-02-01T03:18:02Z","abstract_excerpt":"Over the past few years, Large Language Models (LLMs) have developed rapidly and are widely applied in various domains. However, LLMs face the issue of hallucinations, generating responses that may be unreliable when the models lack relevant knowledge. To be aware of potential hallucinations, uncertainty estimation methods have been introduced, and most of them have confirmed that reliability lies in critical tokens. However, probability-based methods perform poorly in identifying token reliability, limiting their practical utility. In this paper, we reveal that the probability-based method fa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00290","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-01T03:18:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d97c9c26c1a732283fbc61928fbf091a2ada87addb0b2ea53629d1d1a9a8a301","abstract_canon_sha256":"b384ebac23edb0209db4b6fb01b52bcabb50a1582c87d0c45ca7663377160aa8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:41.911766Z","signature_b64":"DTrY6Vn2QLFu9imMj1fFqM6R7b0XiHYIMP/ay/man3S8mXWsyUz+uqraIQcDxIPItT2UlCNe27lx0GXGUYNgAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3fd0e08f8d57c5dba3110b6584e5bd0e42ac56c2a5a95da892b13801c85739e2","last_reissued_at":"2026-07-05T11:00:41.911267Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:41.911267Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Estimating LLM Uncertainty with Evidence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Changqing Zhang, Guangyu Wang, Huan Ma, Jingdong Chen, Joey Tianyi Zhou","submitted_at":"2025-02-01T03:18:02Z","abstract_excerpt":"Over the past few years, Large Language Models (LLMs) have developed rapidly and are widely applied in various domains. However, LLMs face the issue of hallucinations, generating responses that may be unreliable when the models lack relevant knowledge. To be aware of potential hallucinations, uncertainty estimation methods have been introduced, and most of them have confirmed that reliability lies in critical tokens. However, probability-based methods perform poorly in identifying token reliability, limiting their practical utility. In this paper, we reveal that the probability-based method fa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00290","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00290/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00290","created_at":"2026-07-05T11:00:41.911323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00290v5","created_at":"2026-07-05T11:00:41.911323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00290","created_at":"2026-07-05T11:00:41.911323+00:00"},{"alias_kind":"pith_short_12","alias_value":"H7IOBD4NK7C5","created_at":"2026-07-05T11:00:41.911323+00:00"},{"alias_kind":"pith_short_16","alias_value":"H7IOBD4NK7C5XIYR","created_at":"2026-07-05T11:00:41.911323+00:00"},{"alias_kind":"pith_short_8","alias_value":"H7IOBD4N","created_at":"2026-07-05T11:00:41.911323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19509","citing_title":"LLM Doesn't Know What It Doesn't Know: Detecting Epistemic Blind Spots via Cross-Model Attribution Divergence on Clinical Tabular Data","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18043","citing_title":"Uncertainty Quantification for Flow-Based Vision-Language-Action Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17312","citing_title":"Quantifying Consistency in LLM Logical Reasoning via Structural Uncertainty","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09875","citing_title":"Integrating Local and Global Entropy for Uncertainty Quantification in LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23955","citing_title":"From Accuracy to Auditability: A Survey of Determinism in Financial AI Systems","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20772","citing_title":"VIHD: Visual Intervention-based Hallucination Detection for Medical Visual Question Answering","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20772","citing_title":"VIHD: Visual Intervention-based Hallucination Detection for Medical Visual Question Answering","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05777","citing_title":"Estimating the Black-box LLM Uncertainty with Distribution-Aligned Adversarial Distillation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06053","citing_title":"Towards Generation-Efficient Uncertainty Estimation in Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04874","citing_title":"Uncertainty-Aware Exploratory Direct Preference Optimization for Multimodal Large Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17112","citing_title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17821","citing_title":"WebUncertainty: Dual-Level Uncertainty Driven Planning and Reasoning For Autonomous Web Agent","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ","json":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ.json","graph_json":"https://pith.science/api/pith-number/H7IOBD4NK7C5XIYRBNSYJZN5BZ/graph.json","events_json":"https://pith.science/api/pith-number/H7IOBD4NK7C5XIYRBNSYJZN5BZ/events.json","paper":"https://pith.science/paper/H7IOBD4N"},"agent_actions":{"view_html":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ","download_json":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ.json","view_paper":"https://pith.science/paper/H7IOBD4N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00290&json=true","fetch_graph":"https://pith.science/api/pith-number/H7IOBD4NK7C5XIYRBNSYJZN5BZ/graph.json","fetch_events":"https://pith.science/api/pith-number/H7IOBD4NK7C5XIYRBNSYJZN5BZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ/action/storage_attestation","attest_author":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ/action/author_attestation","sign_citation":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ/action/citation_signature","submit_replication":"https://pith.science/pith/H7IOBD4NK7C5XIYRBNSYJZN5BZ/action/replication_record"}},"created_at":"2026-07-05T11:00:41.911323+00:00","updated_at":"2026-07-05T11:00:41.911323+00:00"}