{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:45LPRMJ7TTQRWKNQYIZRHWEBZI","short_pith_number":"pith:45LPRMJ7","schema_version":"1.0","canonical_sha256":"e756f8b13f9ce11b29b0c23313d881ca0fd51f156a763736d4348afee5de8479","source":{"kind":"arxiv","id":"2405.18654","version":3},"attestation_state":"computed","paper":{"title":"Mitigating Object Hallucination in MLLMs via Data-augmented Phrase-level Alignment","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ahmad Beirami, Ali Etemad, Pritam Sarkar, Sayna Ebrahimi, Sercan \\\"O. Ar{\\i}k, Tomas Pfister","submitted_at":"2024-05-28T23:36:00Z","abstract_excerpt":"Despite their significant advancements, Multimodal Large Language Models (MLLMs) often generate factually inaccurate information, referred to as hallucination. In this work, we address object hallucinations in MLLMs, where information is generated about an object not present in the input image. We introduce Data-augmented Phrase-level Alignment (DPA), a novel loss which can be applied to instruction-tuned off-the-shelf MLLMs to mitigate hallucinations, while preserving their general vision-language capabilities. To fine-tune MLLMs with DPA, we first generate a set of `hallucinated' and `correc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.18654","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-28T23:36:00Z","cross_cats_sorted":[],"title_canon_sha256":"2d63bc166586c766c0f90dbc4af350e8d4fd71a739a1d8099ce41f04d741ff78","abstract_canon_sha256":"61fe9ca4c537eaf8b097258a0f236cdb9df32c50b0a3c057ee7bcf8b55e530b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:19.609091Z","signature_b64":"vCVzvV6gfQSxr3p3zmchS7vuqXAMMPCmRnk4vmfZdgWtVwbxc8mL54DM4OqDdaZNSl02pUQxeg8bebteYurODg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e756f8b13f9ce11b29b0c23313d881ca0fd51f156a763736d4348afee5de8479","last_reissued_at":"2026-07-05T10:21:19.608600Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:19.608600Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Object Hallucination in MLLMs via Data-augmented Phrase-level Alignment","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ahmad Beirami, Ali Etemad, Pritam Sarkar, Sayna Ebrahimi, Sercan \\\"O. Ar{\\i}k, Tomas Pfister","submitted_at":"2024-05-28T23:36:00Z","abstract_excerpt":"Despite their significant advancements, Multimodal Large Language Models (MLLMs) often generate factually inaccurate information, referred to as hallucination. In this work, we address object hallucinations in MLLMs, where information is generated about an object not present in the input image. We introduce Data-augmented Phrase-level Alignment (DPA), a novel loss which can be applied to instruction-tuned off-the-shelf MLLMs to mitigate hallucinations, while preserving their general vision-language capabilities. To fine-tune MLLMs with DPA, we first generate a set of `hallucinated' and `correc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.18654","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.18654/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.18654","created_at":"2026-07-05T10:21:19.608660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.18654v3","created_at":"2026-07-05T10:21:19.608660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.18654","created_at":"2026-07-05T10:21:19.608660+00:00"},{"alias_kind":"pith_short_12","alias_value":"45LPRMJ7TTQR","created_at":"2026-07-05T10:21:19.608660+00:00"},{"alias_kind":"pith_short_16","alias_value":"45LPRMJ7TTQRWKNQ","created_at":"2026-07-05T10:21:19.608660+00:00"},{"alias_kind":"pith_short_8","alias_value":"45LPRMJ7","created_at":"2026-07-05T10:21:19.608660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29805","citing_title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29805","citing_title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12455","citing_title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04874","citing_title":"Uncertainty-Aware Exploratory Direct Preference Optimization for Multimodal Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":253,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04641","citing_title":"CAST: Mitigating Object Hallucination in Large Vision-Language Models via Caption-Guided Visual Attention Steering","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI","json":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI.json","graph_json":"https://pith.science/api/pith-number/45LPRMJ7TTQRWKNQYIZRHWEBZI/graph.json","events_json":"https://pith.science/api/pith-number/45LPRMJ7TTQRWKNQYIZRHWEBZI/events.json","paper":"https://pith.science/paper/45LPRMJ7"},"agent_actions":{"view_html":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI","download_json":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI.json","view_paper":"https://pith.science/paper/45LPRMJ7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.18654&json=true","fetch_graph":"https://pith.science/api/pith-number/45LPRMJ7TTQRWKNQYIZRHWEBZI/graph.json","fetch_events":"https://pith.science/api/pith-number/45LPRMJ7TTQRWKNQYIZRHWEBZI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI/action/storage_attestation","attest_author":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI/action/author_attestation","sign_citation":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI/action/citation_signature","submit_replication":"https://pith.science/pith/45LPRMJ7TTQRWKNQYIZRHWEBZI/action/replication_record"}},"created_at":"2026-07-05T10:21:19.608660+00:00","updated_at":"2026-07-05T10:21:19.608660+00:00"}