{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QQIO2SAVZO2364GAFHL6IHQMRL","short_pith_number":"pith:QQIO2SAV","schema_version":"1.0","canonical_sha256":"8410ed4815cbb5bf70c029d7e41e0c8adbe65576a0d2c7299878208e63444867","source":{"kind":"arxiv","id":"2410.04780","version":2},"attestation_state":"computed","paper":{"title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aiwei Liu, Guanyu Zhou, Kun Wang, Xin Zou, Xuming Hu, Yibo Yan","submitted_at":"2024-10-07T06:45:22Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have emerged as a central focus in both industry and academia, but often suffer from biases introduced by visual and language priors, which can lead to multimodal hallucination. These biases arise from the visual encoder and the Large Language Model (LLM) backbone, affecting the attention mechanism responsible for aligning multimodal inputs. Existing decoding-based mitigation methods focus on statistical correlations and overlook the causal relationships between attention mechanisms and model output, limiting their effectiveness in addressing these bias"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04780","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-07T06:45:22Z","cross_cats_sorted":[],"title_canon_sha256":"b55e092688a356a4d5cdf552c8aa523b8702838d0ff374286e795ac29b904b9a","abstract_canon_sha256":"8926c5b639c0edf4d64f1802b484ab248377f4421b2cfe98e8c15ee918f2b827"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:13.424694Z","signature_b64":"rmNvU9pONZNJu/9J2a4pqD3aR88Swt6cDskInEExA+0G++Aovv2mP5YeTffpwnhO+Z4Pm8vBeBGUZlndw8ciAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8410ed4815cbb5bf70c029d7e41e0c8adbe65576a0d2c7299878208e63444867","last_reissued_at":"2026-07-05T10:16:13.424126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:13.424126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Aiwei Liu, Guanyu Zhou, Kun Wang, Xin Zou, Xuming Hu, Yibo Yan","submitted_at":"2024-10-07T06:45:22Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have emerged as a central focus in both industry and academia, but often suffer from biases introduced by visual and language priors, which can lead to multimodal hallucination. These biases arise from the visual encoder and the Large Language Model (LLM) backbone, affecting the attention mechanism responsible for aligning multimodal inputs. Existing decoding-based mitigation methods focus on statistical correlations and overlook the causal relationships between attention mechanisms and model output, limiting their effectiveness in addressing these bias"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04780","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04780/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04780","created_at":"2026-07-05T10:16:13.424179+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04780v2","created_at":"2026-07-05T10:16:13.424179+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04780","created_at":"2026-07-05T10:16:13.424179+00:00"},{"alias_kind":"pith_short_12","alias_value":"QQIO2SAVZO23","created_at":"2026-07-05T10:16:13.424179+00:00"},{"alias_kind":"pith_short_16","alias_value":"QQIO2SAVZO2364GA","created_at":"2026-07-05T10:16:13.424179+00:00"},{"alias_kind":"pith_short_8","alias_value":"QQIO2SAV","created_at":"2026-07-05T10:16:13.424179+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18105","citing_title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2411.16771","citing_title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":265,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18105","citing_title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08003","citing_title":"Rethinking Entropy Allocation in LLM-based ASR: Understanding the Dynamics between Speech Encoders and LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":286,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04641","citing_title":"CAST: Mitigating Object Hallucination in Large Vision-Language Models via Caption-Guided Visual Attention Steering","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL","json":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL.json","graph_json":"https://pith.science/api/pith-number/QQIO2SAVZO2364GAFHL6IHQMRL/graph.json","events_json":"https://pith.science/api/pith-number/QQIO2SAVZO2364GAFHL6IHQMRL/events.json","paper":"https://pith.science/paper/QQIO2SAV"},"agent_actions":{"view_html":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL","download_json":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL.json","view_paper":"https://pith.science/paper/QQIO2SAV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04780&json=true","fetch_graph":"https://pith.science/api/pith-number/QQIO2SAVZO2364GAFHL6IHQMRL/graph.json","fetch_events":"https://pith.science/api/pith-number/QQIO2SAVZO2364GAFHL6IHQMRL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL/action/storage_attestation","attest_author":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL/action/author_attestation","sign_citation":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL/action/citation_signature","submit_replication":"https://pith.science/pith/QQIO2SAVZO2364GAFHL6IHQMRL/action/replication_record"}},"created_at":"2026-07-05T10:16:13.424179+00:00","updated_at":"2026-07-05T10:16:13.424179+00:00"}