{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TY5A5ROPPR57YFHABS4TEBHBLF","short_pith_number":"pith:TY5A5ROP","schema_version":"1.0","canonical_sha256":"9e3a0ec5cf7c7bfc14e00cb93204e15944befbadd3c21b2039156ce461ab19d3","source":{"kind":"arxiv","id":"2404.14233","version":2},"attestation_state":"computed","paper":{"title":"Detecting and Mitigating Hallucination in Large Vision Language Models via Fine-Grained AI Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fangxun Shu, Hao Jiang, Haoyuan Li, Leilei Gan, Linchao Zhu, Wanggui He, Wenyi Xiao, Zhelun Yu, Ziwei Huang","submitted_at":"2024-04-22T14:46:10Z","abstract_excerpt":"The rapidly developing Large Vision Language Models (LVLMs) have shown notable capabilities on a range of multi-modal tasks, but still face the hallucination phenomena where the generated texts do not align with the given contexts, significantly restricting the usages of LVLMs. Most previous work detects and mitigates hallucination at the coarse-grained level or requires expensive annotation (e.g., labeling by proprietary models or human experts). To address these issues, we propose detecting and mitigating hallucinations in LVLMs via fine-grained AI feedback. The basic idea is that we generat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14233","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-22T14:46:10Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"0033d5c0906bf4dcbd5371e2fffd44b118a9d3e3578aa5dfad0ce02d240b1e3f","abstract_canon_sha256":"1d055850bd8cd8499258c83b0e20875245d72006454c32b98a1bc0080c5d5856"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:57:20.062096Z","signature_b64":"Sb40ID0TgZopBT06ArBYD5EYGL3HBSCMiHnByZ3TE3Yq3Z8/ea1hWEcMNY1cv8Eku/+oOeXZkOjYyQGiqNjOAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e3a0ec5cf7c7bfc14e00cb93204e15944befbadd3c21b2039156ce461ab19d3","last_reissued_at":"2026-07-05T09:57:20.061609Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:57:20.061609Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detecting and Mitigating Hallucination in Large Vision Language Models via Fine-Grained AI Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fangxun Shu, Hao Jiang, Haoyuan Li, Leilei Gan, Linchao Zhu, Wanggui He, Wenyi Xiao, Zhelun Yu, Ziwei Huang","submitted_at":"2024-04-22T14:46:10Z","abstract_excerpt":"The rapidly developing Large Vision Language Models (LVLMs) have shown notable capabilities on a range of multi-modal tasks, but still face the hallucination phenomena where the generated texts do not align with the given contexts, significantly restricting the usages of LVLMs. Most previous work detects and mitigates hallucination at the coarse-grained level or requires expensive annotation (e.g., labeling by proprietary models or human experts). To address these issues, we propose detecting and mitigating hallucinations in LVLMs via fine-grained AI feedback. The basic idea is that we generat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14233","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14233","created_at":"2026-07-05T09:57:20.061670+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14233v2","created_at":"2026-07-05T09:57:20.061670+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14233","created_at":"2026-07-05T09:57:20.061670+00:00"},{"alias_kind":"pith_short_12","alias_value":"TY5A5ROPPR57","created_at":"2026-07-05T09:57:20.061670+00:00"},{"alias_kind":"pith_short_16","alias_value":"TY5A5ROPPR57YFHA","created_at":"2026-07-05T09:57:20.061670+00:00"},{"alias_kind":"pith_short_8","alias_value":"TY5A5ROP","created_at":"2026-07-05T09:57:20.061670+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20419","citing_title":"Spectral Query-Key Product Weight Steering for Training-Free VLM Hallucination Mitigation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12455","citing_title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10185","citing_title":"Detecting and Evaluating Medical Hallucinations in Large Vision Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04300","citing_title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10622","citing_title":"Vocabulary Hijacking in LVLMs: Unveiling Critical Attention Heads by Excluding Inert Tokens to Mitigate Hallucination","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":178,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF","json":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF.json","graph_json":"https://pith.science/api/pith-number/TY5A5ROPPR57YFHABS4TEBHBLF/graph.json","events_json":"https://pith.science/api/pith-number/TY5A5ROPPR57YFHABS4TEBHBLF/events.json","paper":"https://pith.science/paper/TY5A5ROP"},"agent_actions":{"view_html":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF","download_json":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF.json","view_paper":"https://pith.science/paper/TY5A5ROP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14233&json=true","fetch_graph":"https://pith.science/api/pith-number/TY5A5ROPPR57YFHABS4TEBHBLF/graph.json","fetch_events":"https://pith.science/api/pith-number/TY5A5ROPPR57YFHABS4TEBHBLF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF/action/storage_attestation","attest_author":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF/action/author_attestation","sign_citation":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF/action/citation_signature","submit_replication":"https://pith.science/pith/TY5A5ROPPR57YFHABS4TEBHBLF/action/replication_record"}},"created_at":"2026-07-05T09:57:20.061670+00:00","updated_at":"2026-07-05T09:57:20.061670+00:00"}