{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JMNA65MWWSMPGRM34Y73QDHWDF","short_pith_number":"pith:JMNA65MW","schema_version":"1.0","canonical_sha256":"4b1a0f7596b498f3459be63fb80cf6196b0133dbee5f075e35175d9545c40d24","source":{"kind":"arxiv","id":"2504.05810","version":2},"attestation_state":"computed","paper":{"title":"PaMi-VDPO: Mitigating Video Hallucinations by Prompt-Aware Multi-Instance Video Preference Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Xu, Jianhua Han, Kui Zhang, Lanqing Hong, Xiaomeng Li, Xinpeng Ding","submitted_at":"2025-04-08T08:41:41Z","abstract_excerpt":"Direct Preference Optimization (DPO) helps reduce hallucinations in Video Multimodal Large Language Models (VLLMs), but its reliance on offline preference data limits adaptability and fails to capture true video-response misalignment. We propose Video Direct Preference Optimization (VDPO), an online preference learning framework that eliminates the need for preference annotation by leveraging video augmentations to generate rejected samples while keeping responses fixed. However, selecting effective augmentations is non-trivial, as some clips may be semantically identical to the original under"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.05810","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-08T08:41:41Z","cross_cats_sorted":[],"title_canon_sha256":"ec8cd203ea335bc2e634048d710751d0bec5d1da1c4ef3e6461f225ee884886d","abstract_canon_sha256":"8703437c2c9da7c02d8bf3254421de3cfb7545a4ad5eff4982822d58da1c27fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:16.809220Z","signature_b64":"xdngpIAPHbmGvUInXEdaoi7pGxqJwWPR4mwxVfmMtr61lz6FEmNhBx4W4f3MoO05iKp9pHELD83nfdD/g8YqCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4b1a0f7596b498f3459be63fb80cf6196b0133dbee5f075e35175d9545c40d24","last_reissued_at":"2026-07-05T10:49:16.808677Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:16.808677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PaMi-VDPO: Mitigating Video Hallucinations by Prompt-Aware Multi-Instance Video Preference Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Xu, Jianhua Han, Kui Zhang, Lanqing Hong, Xiaomeng Li, Xinpeng Ding","submitted_at":"2025-04-08T08:41:41Z","abstract_excerpt":"Direct Preference Optimization (DPO) helps reduce hallucinations in Video Multimodal Large Language Models (VLLMs), but its reliance on offline preference data limits adaptability and fails to capture true video-response misalignment. We propose Video Direct Preference Optimization (VDPO), an online preference learning framework that eliminates the need for preference annotation by leveraging video augmentations to generate rejected samples while keeping responses fixed. However, selecting effective augmentations is non-trivial, as some clips may be semantically identical to the original under"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05810","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.05810/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.05810","created_at":"2026-07-05T10:49:16.808739+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.05810v2","created_at":"2026-07-05T10:49:16.808739+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05810","created_at":"2026-07-05T10:49:16.808739+00:00"},{"alias_kind":"pith_short_12","alias_value":"JMNA65MWWSMP","created_at":"2026-07-05T10:49:16.808739+00:00"},{"alias_kind":"pith_short_16","alias_value":"JMNA65MWWSMPGRM3","created_at":"2026-07-05T10:49:16.808739+00:00"},{"alias_kind":"pith_short_8","alias_value":"JMNA65MW","created_at":"2026-07-05T10:49:16.808739+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11792","citing_title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12582","citing_title":"Relaxing Anchor-Frame Dominance for Mitigating Hallucinations in Video Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12424","citing_title":"Decoding by Perturbation: Mitigating MLLM Hallucinations via Dynamic Textual Perturbation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF","json":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF.json","graph_json":"https://pith.science/api/pith-number/JMNA65MWWSMPGRM34Y73QDHWDF/graph.json","events_json":"https://pith.science/api/pith-number/JMNA65MWWSMPGRM34Y73QDHWDF/events.json","paper":"https://pith.science/paper/JMNA65MW"},"agent_actions":{"view_html":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF","download_json":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF.json","view_paper":"https://pith.science/paper/JMNA65MW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.05810&json=true","fetch_graph":"https://pith.science/api/pith-number/JMNA65MWWSMPGRM34Y73QDHWDF/graph.json","fetch_events":"https://pith.science/api/pith-number/JMNA65MWWSMPGRM34Y73QDHWDF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF/action/storage_attestation","attest_author":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF/action/author_attestation","sign_citation":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF/action/citation_signature","submit_replication":"https://pith.science/pith/JMNA65MWWSMPGRM34Y73QDHWDF/action/replication_record"}},"created_at":"2026-07-05T10:49:16.808739+00:00","updated_at":"2026-07-05T10:49:16.808739+00:00"}