{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FXMADGMTIVBU7INT2OHHBYGWG3","short_pith_number":"pith:FXMADGMT","schema_version":"1.0","canonical_sha256":"2dd801999345434fa1b3d38e70e0d636c109ee32faf645aa09ef27676c09f26e","source":{"kind":"arxiv","id":"2410.11779","version":2},"attestation_state":"computed","paper":{"title":"MLLM can see? Dynamic Correction Decoding for Hallucination Mitigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.MM"],"primary_cat":"cs.CL","authors_text":"Bozhong Tian, Chenxi Wang, Haoming Xu, Huajun Chen, Ningyu Zhang, Shumin Deng, Xiang Chen","submitted_at":"2024-10-15T16:57:44Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) frequently exhibit hallucination phenomena, but the underlying reasons remain poorly understood. In this paper, we present an empirical analysis and find that, although MLLMs incorrectly generate the objects in the final output, they are actually able to recognize visual objects in the preceding layers. We speculate that this may be due to the strong knowledge priors of the language model suppressing the visual information, leading to hallucinations. Motivated by this, we propose a novel dynamic correction decoding method for MLLMs DeCo, which adaptivel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11779","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-15T16:57:44Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG","cs.MM"],"title_canon_sha256":"9e12a8495c400f6a8efe2cfc120ede81bc5c1428762b8cd2047778e8e474fb34","abstract_canon_sha256":"af7b75fc77af265eb8db10ca4acaf3344e8f0a57e1081e9da173df1b117cdde2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:43.346821Z","signature_b64":"HYjKv+QD+n2vYyqsP1u1xwikpB63vLFATzZ83c87CVtcCcfBOfL/o0LcGKye8gs09+tsizboy+B4J/kcP8YHDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dd801999345434fa1b3d38e70e0d636c109ee32faf645aa09ef27676c09f26e","last_reissued_at":"2026-07-05T10:18:43.346325Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:43.346325Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLLM can see? Dynamic Correction Decoding for Hallucination Mitigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG","cs.MM"],"primary_cat":"cs.CL","authors_text":"Bozhong Tian, Chenxi Wang, Haoming Xu, Huajun Chen, Ningyu Zhang, Shumin Deng, Xiang Chen","submitted_at":"2024-10-15T16:57:44Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) frequently exhibit hallucination phenomena, but the underlying reasons remain poorly understood. In this paper, we present an empirical analysis and find that, although MLLMs incorrectly generate the objects in the final output, they are actually able to recognize visual objects in the preceding layers. We speculate that this may be due to the strong knowledge priors of the language model suppressing the visual information, leading to hallucinations. Motivated by this, we propose a novel dynamic correction decoding method for MLLMs DeCo, which adaptivel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11779","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11779/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11779","created_at":"2026-07-05T10:18:43.346389+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11779v2","created_at":"2026-07-05T10:18:43.346389+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11779","created_at":"2026-07-05T10:18:43.346389+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXMADGMTIVBU","created_at":"2026-07-05T10:18:43.346389+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXMADGMTIVBU7INT","created_at":"2026-07-05T10:18:43.346389+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXMADGMT","created_at":"2026-07-05T10:18:43.346389+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27993","citing_title":"Rethinking Visual Neglect: Steering via Context-Preference for MLLM Hallucination Mitigation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17419","citing_title":"EAGLE: Expert-Augmented Attention Guidance for Tuning-Free Industrial Anomaly Detection in Multimodal Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10676","citing_title":"Not Blind but Silenced: Rebalancing Vision and Language via Adversarial Counter-Commonsense Equilibrium","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04874","citing_title":"Uncertainty-Aware Exploratory Direct Preference Optimization for Multimodal Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":162,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12115","citing_title":"HTDC: Hesitation-Triggered Differential Calibration for Mitigating Hallucination in Large Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10071","citing_title":"Spotlight and Shadow: Attention-Guided Dual-Anchor Introspective Decoding for MLLM Hallucination Mitigation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17982","citing_title":"Mitigating Multimodal Hallucination via Phase-wise Self-reward","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":287,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21911","citing_title":"When Prompts Override Vision: Prompt-Induced Hallucinations in LVLMs","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3","json":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3.json","graph_json":"https://pith.science/api/pith-number/FXMADGMTIVBU7INT2OHHBYGWG3/graph.json","events_json":"https://pith.science/api/pith-number/FXMADGMTIVBU7INT2OHHBYGWG3/events.json","paper":"https://pith.science/paper/FXMADGMT"},"agent_actions":{"view_html":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3","download_json":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3.json","view_paper":"https://pith.science/paper/FXMADGMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11779&json=true","fetch_graph":"https://pith.science/api/pith-number/FXMADGMTIVBU7INT2OHHBYGWG3/graph.json","fetch_events":"https://pith.science/api/pith-number/FXMADGMTIVBU7INT2OHHBYGWG3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3/action/storage_attestation","attest_author":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3/action/author_attestation","sign_citation":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3/action/citation_signature","submit_replication":"https://pith.science/pith/FXMADGMTIVBU7INT2OHHBYGWG3/action/replication_record"}},"created_at":"2026-07-05T10:18:43.346389+00:00","updated_at":"2026-07-05T10:18:43.346389+00:00"}