{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:L3EPNW6Y3GZP2GXAJONXFRBJ5S","short_pith_number":"pith:L3EPNW6Y","schema_version":"1.0","canonical_sha256":"5ec8f6dbd8d9b2fd1ae04b9b72c429eca14a46e28e2d35a31f528a370b084c39","source":{"kind":"arxiv","id":"2310.01779","version":3},"attestation_state":"computed","paper":{"title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohan Zhai, Chenfeng Xu, Chunyuan Li, Kurt Keutzer, Manling Li, Sheng Shen, Shijia Yang","submitted_at":"2023-10-03T04:01:27Z","abstract_excerpt":"Current Large Multimodal Models (LMMs) achieve remarkable progress, yet there remains significant uncertainty regarding their ability to accurately apprehend visual details, that is, in performing detailed captioning. To address this, we introduce $\\textit{CCEval}$, a GPT-4 assisted evaluation method for detailed captioning. Interestingly, while LMMs demonstrate minimal object existence hallucination in existing VQA benchmarks, our proposed evaluation reveals continued susceptibility to such hallucinations. In this paper, we make the first attempt to investigate such hallucination from differe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.01779","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-03T04:01:27Z","cross_cats_sorted":[],"title_canon_sha256":"a35ea08d86733385513d7945e7117f066b3e35b195ce0f622f222a462f4312c2","abstract_canon_sha256":"98651232096805510c4e7584c58de2233175efc51ce089c82ff6beb97c93c77e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:02.045179Z","signature_b64":"x/RWO8FYrnlhsE2a2O89YQLaFdOi2MILQfO8LQpR6KrnIK+S+IMZ96+iRKYdQqVaJHJ35I95awW/Txwc3FI4DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ec8f6dbd8d9b2fd1ae04b9b72c429eca14a46e28e2d35a31f528a370b084c39","last_reissued_at":"2026-07-05T08:02:02.044721Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:02.044721Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohan Zhai, Chenfeng Xu, Chunyuan Li, Kurt Keutzer, Manling Li, Sheng Shen, Shijia Yang","submitted_at":"2023-10-03T04:01:27Z","abstract_excerpt":"Current Large Multimodal Models (LMMs) achieve remarkable progress, yet there remains significant uncertainty regarding their ability to accurately apprehend visual details, that is, in performing detailed captioning. To address this, we introduce $\\textit{CCEval}$, a GPT-4 assisted evaluation method for detailed captioning. Interestingly, while LMMs demonstrate minimal object existence hallucination in existing VQA benchmarks, our proposed evaluation reveals continued susceptibility to such hallucinations. In this paper, we make the first attempt to investigate such hallucination from differe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.01779","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.01779/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.01779","created_at":"2026-07-05T08:02:02.044783+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.01779v3","created_at":"2026-07-05T08:02:02.044783+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.01779","created_at":"2026-07-05T08:02:02.044783+00:00"},{"alias_kind":"pith_short_12","alias_value":"L3EPNW6Y3GZP","created_at":"2026-07-05T08:02:02.044783+00:00"},{"alias_kind":"pith_short_16","alias_value":"L3EPNW6Y3GZP2GXA","created_at":"2026-07-05T08:02:02.044783+00:00"},{"alias_kind":"pith_short_8","alias_value":"L3EPNW6Y","created_at":"2026-07-05T08:02:02.044783+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08277","citing_title":"Remember with Confidence: Uncertainty Quantification for Spatio-temporal Memory with Probabilistic Guarantees","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13130","citing_title":"ZINA: Multimodal Fine-grained Hallucination Detection and Editing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21546","citing_title":"Counterfactual Segmentation Reasoning: Diagnosing and Mitigating Pixel-Grounding Hallucination","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21025","citing_title":"CaptionQA: Is Your Caption as Useful as the Image Itself?","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07397","citing_title":"AMBER: An LLM-free Multi-dimensional Benchmark for MLLMs Hallucination Evaluation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":161,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19782","citing_title":"KoALa-Bench: Evaluating Large Audio Language Models on Korean Speech Understanding and Faithfulness","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2402.00253","citing_title":"A Survey on Hallucination in Large Vision-Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10622","citing_title":"Vocabulary Hijacking in LVLMs: Unveiling Critical Attention Heads by Excluding Inert Tokens to Mitigate Hallucination","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S","json":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S.json","graph_json":"https://pith.science/api/pith-number/L3EPNW6Y3GZP2GXAJONXFRBJ5S/graph.json","events_json":"https://pith.science/api/pith-number/L3EPNW6Y3GZP2GXAJONXFRBJ5S/events.json","paper":"https://pith.science/paper/L3EPNW6Y"},"agent_actions":{"view_html":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S","download_json":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S.json","view_paper":"https://pith.science/paper/L3EPNW6Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.01779&json=true","fetch_graph":"https://pith.science/api/pith-number/L3EPNW6Y3GZP2GXAJONXFRBJ5S/graph.json","fetch_events":"https://pith.science/api/pith-number/L3EPNW6Y3GZP2GXAJONXFRBJ5S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S/action/storage_attestation","attest_author":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S/action/author_attestation","sign_citation":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S/action/citation_signature","submit_replication":"https://pith.science/pith/L3EPNW6Y3GZP2GXAJONXFRBJ5S/action/replication_record"}},"created_at":"2026-07-05T08:02:02.044783+00:00","updated_at":"2026-07-05T08:02:02.044783+00:00"}