{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N7AOOSFWJ3TZS2DLTHQJTJGP5H","short_pith_number":"pith:N7AOOSFW","schema_version":"1.0","canonical_sha256":"6fc0e748b64ee799686b99e099a4cfe9c93ea8019e5e041a8cd0e1446cf6ec0a","source":{"kind":"arxiv","id":"2501.09695","version":2},"attestation_state":"computed","paper":{"title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongqi Han, Dongsheng Li, Xufang Luo, Yunjian Xu, Zhihe Yang","submitted_at":"2025-01-16T17:48:03Z","abstract_excerpt":"Hallucination remains a major challenge for Large Vision-Language Models (LVLMs). Direct Preference Optimization (DPO) has gained increasing attention as a simple solution to hallucination issues. It directly learns from constructed preference pairs that reflect the severity of hallucinations in responses to the same prompt and image. Nonetheless, different data construction methods in existing works bring notable performance variations. We identify a crucial factor here: outcomes are largely contingent on whether the constructed data aligns on-policy w.r.t the initial (reference) policy of DP"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09695","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-16T17:48:03Z","cross_cats_sorted":[],"title_canon_sha256":"9217379a84643cc0bc48242caf5ad7b6efbbe8bd34981f445975ab2f266c71db","abstract_canon_sha256":"22218bc1346db979cd50707be92fc055b570a172b31d2452b8e643c4bd56cf2b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:55.898707Z","signature_b64":"ipKRYau2cSlS88uUI0znCCXAgvztZVMzxEfob8Kcg9vBNZ3am2hKOluYYAxvFNnBh6m13PgJsTIilXcN8Vk0AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6fc0e748b64ee799686b99e099a4cfe9c93ea8019e5e041a8cd0e1446cf6ec0a","last_reissued_at":"2026-07-05T10:22:55.898129Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:55.898129Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongqi Han, Dongsheng Li, Xufang Luo, Yunjian Xu, Zhihe Yang","submitted_at":"2025-01-16T17:48:03Z","abstract_excerpt":"Hallucination remains a major challenge for Large Vision-Language Models (LVLMs). Direct Preference Optimization (DPO) has gained increasing attention as a simple solution to hallucination issues. It directly learns from constructed preference pairs that reflect the severity of hallucinations in responses to the same prompt and image. Nonetheless, different data construction methods in existing works bring notable performance variations. We identify a crucial factor here: outcomes are largely contingent on whether the constructed data aligns on-policy w.r.t the initial (reference) policy of DP"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09695","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09695/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09695","created_at":"2026-07-05T10:22:55.898205+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09695v2","created_at":"2026-07-05T10:22:55.898205+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09695","created_at":"2026-07-05T10:22:55.898205+00:00"},{"alias_kind":"pith_short_12","alias_value":"N7AOOSFWJ3TZ","created_at":"2026-07-05T10:22:55.898205+00:00"},{"alias_kind":"pith_short_16","alias_value":"N7AOOSFWJ3TZS2DL","created_at":"2026-07-05T10:22:55.898205+00:00"},{"alias_kind":"pith_short_8","alias_value":"N7AOOSFW","created_at":"2026-07-05T10:22:55.898205+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12590","citing_title":"Analyzing and Improving Fine-grained Preference Optimization in Medical LVLMs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28401","citing_title":"Vision-driven Preference Synthesis for Mitigating Hallucinations in VLMs","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08145","citing_title":"Self-Captioning Multimodal Interaction Tuning: Amplifying Exploitable Redundancies for Robust Vision Language Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":250,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H","json":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H.json","graph_json":"https://pith.science/api/pith-number/N7AOOSFWJ3TZS2DLTHQJTJGP5H/graph.json","events_json":"https://pith.science/api/pith-number/N7AOOSFWJ3TZS2DLTHQJTJGP5H/events.json","paper":"https://pith.science/paper/N7AOOSFW"},"agent_actions":{"view_html":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H","download_json":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H.json","view_paper":"https://pith.science/paper/N7AOOSFW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09695&json=true","fetch_graph":"https://pith.science/api/pith-number/N7AOOSFWJ3TZS2DLTHQJTJGP5H/graph.json","fetch_events":"https://pith.science/api/pith-number/N7AOOSFWJ3TZS2DLTHQJTJGP5H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H/action/storage_attestation","attest_author":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H/action/author_attestation","sign_citation":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H/action/citation_signature","submit_replication":"https://pith.science/pith/N7AOOSFWJ3TZS2DLTHQJTJGP5H/action/replication_record"}},"created_at":"2026-07-05T10:22:55.898205+00:00","updated_at":"2026-07-05T10:22:55.898205+00:00"}