{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3U7HLCYMI24TXPAMPJ26ODVMVS","short_pith_number":"pith:3U7HLCYM","schema_version":"1.0","canonical_sha256":"dd3e758b0c46b93bbc0c7a75e70eacaca1de704e84a27c3bf942bc3b99f1d821","source":{"kind":"arxiv","id":"2501.16629","version":1},"attestation_state":"computed","paper":{"title":"CHiP: Cross-modal Hierarchical Direct Preference Optimization for Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Bryan Hooi, Hao Fei, Jinlan Fu, See-kiong Ng, Shenzhen Huangfu, Xiaoyu Shen, Xipeng Qiu","submitted_at":"2025-01-28T02:05:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) still struggle with hallucinations despite their impressive capabilities. Recent studies have attempted to mitigate this by applying Direct Preference Optimization (DPO) to multimodal scenarios using preference pairs from text-based responses. However, our analysis of representation distributions reveals that multimodal DPO struggles to align image and text representations and to distinguish between hallucinated and non-hallucinated descriptions. To address these challenges, in this work, we propose a Cross-modal Hierarchical Direct Preference Optimizat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16629","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-01-28T02:05:38Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"b25500d928c7181a629122914ccca492f7e0160c60e7bc40bb5bc8984f47dbde","abstract_canon_sha256":"0ab00d08acb74defbf11f823887c1b105570b3628d9411314b7ffb24e69f967e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:16.105075Z","signature_b64":"8WWckZ2RjFw8ZsMSQee6qEW0JZDwe5MTe6c4uQwhnMCexIPb3E9IxtK5IsB62a12IZRy2+4KAHDI9fx1qtSeAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dd3e758b0c46b93bbc0c7a75e70eacaca1de704e84a27c3bf942bc3b99f1d821","last_reissued_at":"2026-07-05T10:06:16.104624Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:16.104624Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CHiP: Cross-modal Hierarchical Direct Preference Optimization for Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Bryan Hooi, Hao Fei, Jinlan Fu, See-kiong Ng, Shenzhen Huangfu, Xiaoyu Shen, Xipeng Qiu","submitted_at":"2025-01-28T02:05:38Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) still struggle with hallucinations despite their impressive capabilities. Recent studies have attempted to mitigate this by applying Direct Preference Optimization (DPO) to multimodal scenarios using preference pairs from text-based responses. However, our analysis of representation distributions reveals that multimodal DPO struggles to align image and text representations and to distinguish between hallucinated and non-hallucinated descriptions. To address these challenges, in this work, we propose a Cross-modal Hierarchical Direct Preference Optimizat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16629","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16629/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16629","created_at":"2026-07-05T10:06:16.104687+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16629v1","created_at":"2026-07-05T10:06:16.104687+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16629","created_at":"2026-07-05T10:06:16.104687+00:00"},{"alias_kind":"pith_short_12","alias_value":"3U7HLCYMI24T","created_at":"2026-07-05T10:06:16.104687+00:00"},{"alias_kind":"pith_short_16","alias_value":"3U7HLCYMI24TXPAM","created_at":"2026-07-05T10:06:16.104687+00:00"},{"alias_kind":"pith_short_8","alias_value":"3U7HLCYM","created_at":"2026-07-05T10:06:16.104687+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12590","citing_title":"Analyzing and Improving Fine-grained Preference Optimization in Medical LVLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10622","citing_title":"Vocabulary Hijacking in LVLMs: Unveiling Critical Attention Heads by Excluding Inert Tokens to Mitigate Hallucination","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17982","citing_title":"Mitigating Multimodal Hallucination via Phase-wise Self-reward","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20366","citing_title":"Mitigating Hallucinations in Large Vision-Language Models without Performance Degradation","ref_index":170,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS","json":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS.json","graph_json":"https://pith.science/api/pith-number/3U7HLCYMI24TXPAMPJ26ODVMVS/graph.json","events_json":"https://pith.science/api/pith-number/3U7HLCYMI24TXPAMPJ26ODVMVS/events.json","paper":"https://pith.science/paper/3U7HLCYM"},"agent_actions":{"view_html":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS","download_json":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS.json","view_paper":"https://pith.science/paper/3U7HLCYM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16629&json=true","fetch_graph":"https://pith.science/api/pith-number/3U7HLCYMI24TXPAMPJ26ODVMVS/graph.json","fetch_events":"https://pith.science/api/pith-number/3U7HLCYMI24TXPAMPJ26ODVMVS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS/action/storage_attestation","attest_author":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS/action/author_attestation","sign_citation":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS/action/citation_signature","submit_replication":"https://pith.science/pith/3U7HLCYMI24TXPAMPJ26ODVMVS/action/replication_record"}},"created_at":"2026-07-05T10:06:16.104687+00:00","updated_at":"2026-07-05T10:06:16.104687+00:00"}