{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UPQ23KJTDYRNHMKHEMZOHXIJSH","short_pith_number":"pith:UPQ23KJT","schema_version":"1.0","canonical_sha256":"a3e1ada9331e22d3b1472332e3dd0991f3e5f76cda119d5475841a302d534e4c","source":{"kind":"arxiv","id":"2406.11839","version":2},"attestation_state":"computed","paper":{"title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fei Wang, Hoifung Poon, James Y. Huang, Muhao Chen, Nan Xu, Sheng Zhang, Wenxuan Zhou","submitted_at":"2024-06-17T17:59:58Z","abstract_excerpt":"Direct preference optimization (DPO) has shown to be an effective method for large language model (LLM) alignment. Recent works have attempted to apply DPO to multimodal scenarios but have found it challenging to achieve consistent improvement. Through a comparative experiment, we identify the unconditional preference problem in multimodal preference optimization, where the model overlooks the image condition. To address this problem, we propose mDPO, a multimodal DPO objective that prevents the over-prioritization of language-only preferences by also optimizing image preference. Moreover, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11839","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-17T17:59:58Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"8e457f448976e281734bc880ee3a876ec4cc87ae14194068ef78b273982bb071","abstract_canon_sha256":"f3c785f35f5eb23c6d209ca96f5d07a020d3fac1aa1d44b7a5d2b4cfc5a8a42f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:50.174806Z","signature_b64":"T7W7WEv5ZjChLuwM6/v0V4TCVzHmduu80srDhZul0C2UM7OX9BYV+2Js8gkIjs/uU6NrFwpyZaCW3zwVCIl3CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a3e1ada9331e22d3b1472332e3dd0991f3e5f76cda119d5475841a302d534e4c","last_reissued_at":"2026-07-05T09:16:50.174308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:50.174308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fei Wang, Hoifung Poon, James Y. Huang, Muhao Chen, Nan Xu, Sheng Zhang, Wenxuan Zhou","submitted_at":"2024-06-17T17:59:58Z","abstract_excerpt":"Direct preference optimization (DPO) has shown to be an effective method for large language model (LLM) alignment. Recent works have attempted to apply DPO to multimodal scenarios but have found it challenging to achieve consistent improvement. Through a comparative experiment, we identify the unconditional preference problem in multimodal preference optimization, where the model overlooks the image condition. To address this problem, we propose mDPO, a multimodal DPO objective that prevents the over-prioritization of language-only preferences by also optimizing image preference. Moreover, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11839","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11839/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11839","created_at":"2026-07-05T09:16:50.174381+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11839v2","created_at":"2026-07-05T09:16:50.174381+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11839","created_at":"2026-07-05T09:16:50.174381+00:00"},{"alias_kind":"pith_short_12","alias_value":"UPQ23KJTDYRN","created_at":"2026-07-05T09:16:50.174381+00:00"},{"alias_kind":"pith_short_16","alias_value":"UPQ23KJTDYRNHMKH","created_at":"2026-07-05T09:16:50.174381+00:00"},{"alias_kind":"pith_short_8","alias_value":"UPQ23KJT","created_at":"2026-07-05T09:16:50.174381+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24636","citing_title":"CineCap: Structured Reasoning with Spatio-Temporal Anchors for Cinematographic Video Captioning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00247","citing_title":"Adaptive Perturbation Selection for Contrastive Audio Decoding","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07647","citing_title":"Steer Where It Matters: Token-Level Visual-Sensitivity Steering for LVLMs Hallucination Mitigation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03376","citing_title":"P$^2$-DPO: Grounding Hallucination in Perceptual Processing via Calibration Direct Preference Optimization","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":193,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH","json":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH.json","graph_json":"https://pith.science/api/pith-number/UPQ23KJTDYRNHMKHEMZOHXIJSH/graph.json","events_json":"https://pith.science/api/pith-number/UPQ23KJTDYRNHMKHEMZOHXIJSH/events.json","paper":"https://pith.science/paper/UPQ23KJT"},"agent_actions":{"view_html":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH","download_json":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH.json","view_paper":"https://pith.science/paper/UPQ23KJT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11839&json=true","fetch_graph":"https://pith.science/api/pith-number/UPQ23KJTDYRNHMKHEMZOHXIJSH/graph.json","fetch_events":"https://pith.science/api/pith-number/UPQ23KJTDYRNHMKHEMZOHXIJSH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH/action/storage_attestation","attest_author":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH/action/author_attestation","sign_citation":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH/action/citation_signature","submit_replication":"https://pith.science/pith/UPQ23KJTDYRNHMKHEMZOHXIJSH/action/replication_record"}},"created_at":"2026-07-05T09:16:50.174381+00:00","updated_at":"2026-07-05T09:16:50.174381+00:00"}