{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K7NA5V4VOV2E74WULNJMUUJVUJ","short_pith_number":"pith:K7NA5V4V","schema_version":"1.0","canonical_sha256":"57da0ed79575744ff2d45b52ca5135a271aa7bf5a1bb770652ef3e108e5c441c","source":{"kind":"arxiv","id":"2411.02712","version":1},"attestation_state":"computed","paper":{"title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Guanzhen Li, Min-Yen Kan, Xiao Xu, Yuxi Xie","submitted_at":"2024-11-05T01:24:37Z","abstract_excerpt":"Large vision-language models (LVLMs) suffer from hallucination, resulting in misalignment between the output textual response and the input visual content. Recent research indicates that the over-reliance on the Large Language Model (LLM) backbone, as one cause of the LVLM hallucination, inherently introduces bias from language priors, leading to insufficient context attention to the visual inputs.\n  We tackle this issue of hallucination by mitigating such over-reliance through preference learning. We propose Vision-guided Direct Preference Optimization (V-DPO) to enhance visual context learni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02712","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-05T01:24:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1472a6c00544cdcfdc08d20e345f4450e8cdc651309a27b0dae42bce561f52f8","abstract_canon_sha256":"9c55fa359f4ffda117b25d4234796fb915fe30eebad837422d08f75f57e41638"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:16.741474Z","signature_b64":"AnexjhpWwViIxjjMwXrjI1oI4dv7X9TN1nCYNkocVcD5ZSD3D9SmbJNQVtuoagP9xOoS0DGfbGDJHmeELmMuAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57da0ed79575744ff2d45b52ca5135a271aa7bf5a1bb770652ef3e108e5c441c","last_reissued_at":"2026-07-05T09:31:16.740994Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:16.740994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Guanzhen Li, Min-Yen Kan, Xiao Xu, Yuxi Xie","submitted_at":"2024-11-05T01:24:37Z","abstract_excerpt":"Large vision-language models (LVLMs) suffer from hallucination, resulting in misalignment between the output textual response and the input visual content. Recent research indicates that the over-reliance on the Large Language Model (LLM) backbone, as one cause of the LVLM hallucination, inherently introduces bias from language priors, leading to insufficient context attention to the visual inputs.\n  We tackle this issue of hallucination by mitigating such over-reliance through preference learning. We propose Vision-guided Direct Preference Optimization (V-DPO) to enhance visual context learni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02712","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02712/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02712","created_at":"2026-07-05T09:31:16.741052+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02712v1","created_at":"2026-07-05T09:31:16.741052+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02712","created_at":"2026-07-05T09:31:16.741052+00:00"},{"alias_kind":"pith_short_12","alias_value":"K7NA5V4VOV2E","created_at":"2026-07-05T09:31:16.741052+00:00"},{"alias_kind":"pith_short_16","alias_value":"K7NA5V4VOV2E74WU","created_at":"2026-07-05T09:31:16.741052+00:00"},{"alias_kind":"pith_short_8","alias_value":"K7NA5V4V","created_at":"2026-07-05T09:31:16.741052+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07647","citing_title":"Steer Where It Matters: Token-Level Visual-Sensitivity Steering for LVLMs Hallucination Mitigation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03376","citing_title":"P$^2$-DPO: Grounding Hallucination in Perceptual Processing via Calibration Direct Preference Optimization","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00323","citing_title":"Online Self-Calibration Against Hallucination in Vision-Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":180,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14129","citing_title":"Don't Let the Video Speak: Audio-Contrastive Preference Optimization for Audio-Visual Language Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21911","citing_title":"When Prompts Override Vision: Prompt-Induced Hallucinations in LVLMs","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ","json":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ.json","graph_json":"https://pith.science/api/pith-number/K7NA5V4VOV2E74WULNJMUUJVUJ/graph.json","events_json":"https://pith.science/api/pith-number/K7NA5V4VOV2E74WULNJMUUJVUJ/events.json","paper":"https://pith.science/paper/K7NA5V4V"},"agent_actions":{"view_html":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ","download_json":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ.json","view_paper":"https://pith.science/paper/K7NA5V4V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02712&json=true","fetch_graph":"https://pith.science/api/pith-number/K7NA5V4VOV2E74WULNJMUUJVUJ/graph.json","fetch_events":"https://pith.science/api/pith-number/K7NA5V4VOV2E74WULNJMUUJVUJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ/action/storage_attestation","attest_author":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ/action/author_attestation","sign_citation":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ/action/citation_signature","submit_replication":"https://pith.science/pith/K7NA5V4VOV2E74WULNJMUUJVUJ/action/replication_record"}},"created_at":"2026-07-05T09:31:16.741052+00:00","updated_at":"2026-07-05T09:31:16.741052+00:00"}