{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UGYN4LX2E2W5CYLMCN3RCTADNP","short_pith_number":"pith:UGYN4LX2","schema_version":"1.0","canonical_sha256":"a1b0de2efa26add1616c1377114c036bc302d0ec53ac292fa66129de9f06a917","source":{"kind":"arxiv","id":"2505.16652","version":2},"attestation_state":"computed","paper":{"title":"Seeing Far and Clearly: Mitigating Hallucinations in MLLMs with Attention Causal Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chengzhi Liu, Feilong Tang, Imran Razzak, Jionglong Su, Ming Hu, Minquan Lin, Xuelian Cheng, Yifan Peng, Zelin Peng, Zhiwei Yang, Zhongxing Xu, Zongyuan Ge","submitted_at":"2025-05-22T13:19:57Z","abstract_excerpt":"Recent advancements in multimodal large language models (MLLMs) have significantly improved performance in visual question answering. However, they often suffer from hallucinations. In this work, hallucinations are categorized into two main types: initial hallucinations and snowball hallucinations. We argue that adequate contextual information can be extracted directly from the token interaction process. Inspired by causal inference in the decoding strategy, we propose to leverage causal masks to establish information propagation between multimodal tokens. The hypothesis is that insufficient i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16652","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-22T13:19:57Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"da304c21321639005ed7d4f734c21bb7f2f00bf860590bebc343076ac0e5cdbf","abstract_canon_sha256":"a90b00dcdd69e901aad250fcdcf858555264e80b0de78fe5bb5b01764278fa22"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:45.647355Z","signature_b64":"lzV6p+zj3uRHC8POikhnbkXZsd2GAGWi3zeQtfNCVhM0tQhcF4uSB8A5jf6AUExvy7QTbzjS2KohTPKbjKpgCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1b0de2efa26add1616c1377114c036bc302d0ec53ac292fa66129de9f06a917","last_reissued_at":"2026-07-05T11:17:45.646865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:45.646865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Seeing Far and Clearly: Mitigating Hallucinations in MLLMs with Attention Causal Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chengzhi Liu, Feilong Tang, Imran Razzak, Jionglong Su, Ming Hu, Minquan Lin, Xuelian Cheng, Yifan Peng, Zelin Peng, Zhiwei Yang, Zhongxing Xu, Zongyuan Ge","submitted_at":"2025-05-22T13:19:57Z","abstract_excerpt":"Recent advancements in multimodal large language models (MLLMs) have significantly improved performance in visual question answering. However, they often suffer from hallucinations. In this work, hallucinations are categorized into two main types: initial hallucinations and snowball hallucinations. We argue that adequate contextual information can be extracted directly from the token interaction process. Inspired by causal inference in the decoding strategy, we propose to leverage causal masks to establish information propagation between multimodal tokens. The hypothesis is that insufficient i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16652","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16652/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16652","created_at":"2026-07-05T11:17:45.646922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16652v2","created_at":"2026-07-05T11:17:45.646922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16652","created_at":"2026-07-05T11:17:45.646922+00:00"},{"alias_kind":"pith_short_12","alias_value":"UGYN4LX2E2W5","created_at":"2026-07-05T11:17:45.646922+00:00"},{"alias_kind":"pith_short_16","alias_value":"UGYN4LX2E2W5CYLM","created_at":"2026-07-05T11:17:45.646922+00:00"},{"alias_kind":"pith_short_8","alias_value":"UGYN4LX2","created_at":"2026-07-05T11:17:45.646922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":274,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP","json":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP.json","graph_json":"https://pith.science/api/pith-number/UGYN4LX2E2W5CYLMCN3RCTADNP/graph.json","events_json":"https://pith.science/api/pith-number/UGYN4LX2E2W5CYLMCN3RCTADNP/events.json","paper":"https://pith.science/paper/UGYN4LX2"},"agent_actions":{"view_html":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP","download_json":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP.json","view_paper":"https://pith.science/paper/UGYN4LX2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16652&json=true","fetch_graph":"https://pith.science/api/pith-number/UGYN4LX2E2W5CYLMCN3RCTADNP/graph.json","fetch_events":"https://pith.science/api/pith-number/UGYN4LX2E2W5CYLMCN3RCTADNP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP/action/storage_attestation","attest_author":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP/action/author_attestation","sign_citation":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP/action/citation_signature","submit_replication":"https://pith.science/pith/UGYN4LX2E2W5CYLMCN3RCTADNP/action/replication_record"}},"created_at":"2026-07-05T11:17:45.646922+00:00","updated_at":"2026-07-05T11:17:45.646922+00:00"}