{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4O35KQMFPA3BYGLGEDZQZZ7DND","short_pith_number":"pith:4O35KQMF","schema_version":"1.0","canonical_sha256":"e3b7d5418578361c196620f30ce7e368cc2c6eae0f8a592d6295951831d0f4fe","source":{"kind":"arxiv","id":"2404.01258","version":2},"attestation_state":"computed","paper":{"title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Hauptmann, Chunyuan Li, Di Fu, Keyang Xu, Liangke Gui, Ruohong Zhang, Yihao Feng, Yiming Yang, Yonatan Bisk, Yuanhan Zhang, Zhiqing Sun","submitted_at":"2024-04-01T17:28:16Z","abstract_excerpt":"Preference modeling techniques, such as direct preference optimization (DPO), has shown effective in enhancing the generalization abilities of large language model (LLM). However, in tasks involving video instruction-following, providing informative feedback, especially for detecting hallucinations in generated responses, remains a significant challenge. Previous studies have explored using large large multimodal models (LMMs) as reward models to guide preference modeling, but their ability to accurately assess the factuality of generated responses compared to corresponding videos has not been"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.01258","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-01T17:28:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c19612b1332b091b80562dfca4414390efaf1ac857eaf2efd33d38cd7261d90f","abstract_canon_sha256":"9b4d08740e608c1227682699dfd5584dadf097a8a4001e944aca673aaaf1ba3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:03:30.724361Z","signature_b64":"tYZxD9tkK+4jnNKV7vCrAPM6tTfzD7GwAirZYgS8KZk1Ms9vRYcPPCNN9y/9KnEC7J1UQWYedbiJmTdt4rkQCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e3b7d5418578361c196620f30ce7e368cc2c6eae0f8a592d6295951831d0f4fe","last_reissued_at":"2026-07-05T08:03:30.723718Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:03:30.723718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Hauptmann, Chunyuan Li, Di Fu, Keyang Xu, Liangke Gui, Ruohong Zhang, Yihao Feng, Yiming Yang, Yonatan Bisk, Yuanhan Zhang, Zhiqing Sun","submitted_at":"2024-04-01T17:28:16Z","abstract_excerpt":"Preference modeling techniques, such as direct preference optimization (DPO), has shown effective in enhancing the generalization abilities of large language model (LLM). However, in tasks involving video instruction-following, providing informative feedback, especially for detecting hallucinations in generated responses, remains a significant challenge. Previous studies have explored using large large multimodal models (LMMs) as reward models to guide preference modeling, but their ability to accurately assess the factuality of generated responses compared to corresponding videos has not been"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.01258","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.01258/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.01258","created_at":"2026-07-05T08:03:30.723794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.01258v2","created_at":"2026-07-05T08:03:30.723794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.01258","created_at":"2026-07-05T08:03:30.723794+00:00"},{"alias_kind":"pith_short_12","alias_value":"4O35KQMFPA3B","created_at":"2026-07-05T08:03:30.723794+00:00"},{"alias_kind":"pith_short_16","alias_value":"4O35KQMFPA3BYGLG","created_at":"2026-07-05T08:03:30.723794+00:00"},{"alias_kind":"pith_short_8","alias_value":"4O35KQMF","created_at":"2026-07-05T08:03:30.723794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":209,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11792","citing_title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10759","citing_title":"miniReranker: Efficient Multimodal Reranking through Visual Cache Reuse and Interaction Sparsity","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03376","citing_title":"P$^2$-DPO: Grounding Hallucination in Perceptual Processing via Calibration Direct Preference Optimization","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04468","citing_title":"NVILA: Efficient Frontier Visual Language Models","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05067","citing_title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22819","citing_title":"Cambrian-P: Pose-Grounded Video Understanding","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2408.04840","citing_title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","ref_index":268,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04590","citing_title":"VLM2Vec-V2: Advancing Multimodal Embedding for Videos, Images, and Visual Documents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00574","citing_title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2503.05236","citing_title":"Unified Reward Model for Multimodal Understanding and Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02467","citing_title":"VERTIGO: Visual Preference Optimization for Cinematic Camera Trajectory Generation","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27002","citing_title":"Membership Inference Attacks Against Video Large Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21718","citing_title":"Building a Precise Video Language with Human-AI Oversight","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07895","citing_title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2408.03326","citing_title":"LLaVA-OneVision: Easy Visual Task Transfer","ref_index":167,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND","json":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND.json","graph_json":"https://pith.science/api/pith-number/4O35KQMFPA3BYGLGEDZQZZ7DND/graph.json","events_json":"https://pith.science/api/pith-number/4O35KQMFPA3BYGLGEDZQZZ7DND/events.json","paper":"https://pith.science/paper/4O35KQMF"},"agent_actions":{"view_html":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND","download_json":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND.json","view_paper":"https://pith.science/paper/4O35KQMF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.01258&json=true","fetch_graph":"https://pith.science/api/pith-number/4O35KQMFPA3BYGLGEDZQZZ7DND/graph.json","fetch_events":"https://pith.science/api/pith-number/4O35KQMFPA3BYGLGEDZQZZ7DND/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND/action/storage_attestation","attest_author":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND/action/author_attestation","sign_citation":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND/action/citation_signature","submit_replication":"https://pith.science/pith/4O35KQMFPA3BYGLGEDZQZZ7DND/action/replication_record"}},"created_at":"2026-07-05T08:03:30.723794+00:00","updated_at":"2026-07-05T08:03:30.723794+00:00"}