{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CCLNX4XVR3VT6TE54XK274M7YI","short_pith_number":"pith:CCLNX4XV","schema_version":"1.0","canonical_sha256":"1096dbf2f58eeb3f4c9de5d5aff19fc2338be3c913fbc1125c23b7c1d39f4c59","source":{"kind":"arxiv","id":"2308.15126","version":3},"attestation_state":"computed","paper":{"title":"Evaluation and Analysis of Hallucination in Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Chenlin Zhao, Guohai Xu, Haiyang Xu, Haoyu Tang, Jihua Zhu, Jitao Sang, Ji Zhang, Junyang Wang, Ming Yan, Pengcheng Shi, Qinghao Ye, Yiyang Zhou","submitted_at":"2023-08-29T08:51:24Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have recently achieved remarkable success. However, LVLMs are still plagued by the hallucination problem, which limits the practicality in many scenarios. Hallucination refers to the information of LVLMs' responses that does not exist in the visual input, which poses potential risks of substantial consequences. There has been limited work studying hallucination evaluation in LVLMs. In this paper, we propose Hallucination Evaluation based on Large Language Models (HaELM), an LLM-based hallucination evaluation framework. HaELM achieves an approximate 95% perf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.15126","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-29T08:51:24Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"f03a949d8ecff44b4dab0f81388939fff4d350d0b6b1b5c510217ab44fc53340","abstract_canon_sha256":"f66b39dfe997855147994ecaa568d2de7666365bd47f431a793723e4c4db8b51"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:59.259810Z","signature_b64":"TsOMjGCw1yORksqPZcVDInttIfJMlSu3tqlAvAn/oArW4iuwh4tq1BUH8zAByNDFgQQ6utmjTlKC1bbPup2yDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1096dbf2f58eeb3f4c9de5d5aff19fc2338be3c913fbc1125c23b7c1d39f4c59","last_reissued_at":"2026-07-05T06:58:59.259273Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:59.259273Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation and Analysis of Hallucination in Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Chenlin Zhao, Guohai Xu, Haiyang Xu, Haoyu Tang, Jihua Zhu, Jitao Sang, Ji Zhang, Junyang Wang, Ming Yan, Pengcheng Shi, Qinghao Ye, Yiyang Zhou","submitted_at":"2023-08-29T08:51:24Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have recently achieved remarkable success. However, LVLMs are still plagued by the hallucination problem, which limits the practicality in many scenarios. Hallucination refers to the information of LVLMs' responses that does not exist in the visual input, which poses potential risks of substantial consequences. There has been limited work studying hallucination evaluation in LVLMs. In this paper, we propose Hallucination Evaluation based on Large Language Models (HaELM), an LLM-based hallucination evaluation framework. HaELM achieves an approximate 95% perf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.15126","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.15126/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.15126","created_at":"2026-07-05T06:58:59.259348+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.15126v3","created_at":"2026-07-05T06:58:59.259348+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.15126","created_at":"2026-07-05T06:58:59.259348+00:00"},{"alias_kind":"pith_short_12","alias_value":"CCLNX4XVR3VT","created_at":"2026-07-05T06:58:59.259348+00:00"},{"alias_kind":"pith_short_16","alias_value":"CCLNX4XVR3VT6TE5","created_at":"2026-07-05T06:58:59.259348+00:00"},{"alias_kind":"pith_short_8","alias_value":"CCLNX4XV","created_at":"2026-07-05T06:58:59.259348+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26923","citing_title":"GAVEL: Grounded Caption Error Verification and Localization","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24602","citing_title":"Correcting Visual Blur Induced by Attention Distraction to Reduce Hallucinations: Algorithm and Theory","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10185","citing_title":"Detecting and Evaluating Medical Hallucinations in Large Vision Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2310.00754","citing_title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08392","citing_title":"ST-BiBench: Benchmarking Multi-Stream Multimodal Coordination in Bimanual Embodied Tasks for MLLMs","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16502","citing_title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2402.00253","citing_title":"A Survey on Hallucination in Large Vision-Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11808","citing_title":"Mitigating Action-Relation Hallucinations in LVLMs via Relation-aware Visual Enhancement","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05623","citing_title":"DetailVerifyBench: A Benchmark for Dense Hallucination Localization in Long Image Captions","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18803","citing_title":"LLM-as-Judge Framework for Evaluating Tone-Induced Hallucination in Vision-Language Models","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI","json":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI.json","graph_json":"https://pith.science/api/pith-number/CCLNX4XVR3VT6TE54XK274M7YI/graph.json","events_json":"https://pith.science/api/pith-number/CCLNX4XVR3VT6TE54XK274M7YI/events.json","paper":"https://pith.science/paper/CCLNX4XV"},"agent_actions":{"view_html":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI","download_json":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI.json","view_paper":"https://pith.science/paper/CCLNX4XV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.15126&json=true","fetch_graph":"https://pith.science/api/pith-number/CCLNX4XVR3VT6TE54XK274M7YI/graph.json","fetch_events":"https://pith.science/api/pith-number/CCLNX4XVR3VT6TE54XK274M7YI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI/action/storage_attestation","attest_author":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI/action/author_attestation","sign_citation":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI/action/citation_signature","submit_replication":"https://pith.science/pith/CCLNX4XVR3VT6TE54XK274M7YI/action/replication_record"}},"created_at":"2026-07-05T06:58:59.259348+00:00","updated_at":"2026-07-05T06:58:59.259348+00:00"}