{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LK4B6D5AYVJR4SXBP3AV32TRBT","short_pith_number":"pith:LK4B6D5A","schema_version":"1.0","canonical_sha256":"5ab81f0fa0c5531e4ae17ec15dea710cea95155f43dd9ce1659ec9b902ee6778","source":{"kind":"arxiv","id":"2506.00245","version":1},"attestation_state":"computed","paper":{"title":"Beyond Semantic Entropy: Boosting LLM Uncertainty Quantification with Pairwise Semantic Similarity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ali Payani, Baharan Mirzasoleiman, Dang Nguyen","submitted_at":"2025-05-30T21:21:05Z","abstract_excerpt":"Hallucination in large language models (LLMs) can be detected by assessing the uncertainty of model outputs, typically measured using entropy. Semantic entropy (SE) enhances traditional entropy estimation by quantifying uncertainty at the semantic cluster level. However, as modern LLMs generate longer one-sentence responses, SE becomes less effective because it overlooks two crucial factors: intra-cluster similarity (the spread within a cluster) and inter-cluster similarity (the distance between clusters). To address these limitations, we propose a simple black-box uncertainty quantification m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00245","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T21:21:05Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"58d877c3bec21d40a0f753eb2a65cf93a25205a6972a390c61d3cf56bc9f8ee4","abstract_canon_sha256":"09ba47f14d38fe9d1ce6d6fc2d4fe86fd93b84fb79e8409e823e33ec9438f73f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:31.459691Z","signature_b64":"5A0GN1MD8Es3fGGFZzT4n3yjEl28U+z0K2rhIwUzIq1K9zPjyILR0ny0Hv9VnkFFx1q4Gs2bkGpwON0PRSTnBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ab81f0fa0c5531e4ae17ec15dea710cea95155f43dd9ce1659ec9b902ee6778","last_reissued_at":"2026-07-05T11:13:31.459244Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:31.459244Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Semantic Entropy: Boosting LLM Uncertainty Quantification with Pairwise Semantic Similarity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ali Payani, Baharan Mirzasoleiman, Dang Nguyen","submitted_at":"2025-05-30T21:21:05Z","abstract_excerpt":"Hallucination in large language models (LLMs) can be detected by assessing the uncertainty of model outputs, typically measured using entropy. Semantic entropy (SE) enhances traditional entropy estimation by quantifying uncertainty at the semantic cluster level. However, as modern LLMs generate longer one-sentence responses, SE becomes less effective because it overlooks two crucial factors: intra-cluster similarity (the spread within a cluster) and inter-cluster similarity (the distance between clusters). To address these limitations, we propose a simple black-box uncertainty quantification m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00245","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00245/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00245","created_at":"2026-07-05T11:13:31.459300+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00245v1","created_at":"2026-07-05T11:13:31.459300+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00245","created_at":"2026-07-05T11:13:31.459300+00:00"},{"alias_kind":"pith_short_12","alias_value":"LK4B6D5AYVJR","created_at":"2026-07-05T11:13:31.459300+00:00"},{"alias_kind":"pith_short_16","alias_value":"LK4B6D5AYVJR4SXB","created_at":"2026-07-05T11:13:31.459300+00:00"},{"alias_kind":"pith_short_8","alias_value":"LK4B6D5A","created_at":"2026-07-05T11:13:31.459300+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19220","citing_title":"Position: Uncertainty Quantification in LLMs is Just Unsupervised Clustering","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01458","citing_title":"When to Trust the Answer: Question-Aligned Semantic Nearest Neighbor Entropy for Safer Surgical VQA","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27098","citing_title":"Ensemble-Based Uncertainty Estimation for Code Correctness Estimation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT","json":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT.json","graph_json":"https://pith.science/api/pith-number/LK4B6D5AYVJR4SXBP3AV32TRBT/graph.json","events_json":"https://pith.science/api/pith-number/LK4B6D5AYVJR4SXBP3AV32TRBT/events.json","paper":"https://pith.science/paper/LK4B6D5A"},"agent_actions":{"view_html":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT","download_json":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT.json","view_paper":"https://pith.science/paper/LK4B6D5A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00245&json=true","fetch_graph":"https://pith.science/api/pith-number/LK4B6D5AYVJR4SXBP3AV32TRBT/graph.json","fetch_events":"https://pith.science/api/pith-number/LK4B6D5AYVJR4SXBP3AV32TRBT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT/action/storage_attestation","attest_author":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT/action/author_attestation","sign_citation":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT/action/citation_signature","submit_replication":"https://pith.science/pith/LK4B6D5AYVJR4SXBP3AV32TRBT/action/replication_record"}},"created_at":"2026-07-05T11:13:31.459300+00:00","updated_at":"2026-07-05T11:13:31.459300+00:00"}