{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KTWN7CGZLJ53XJX7WLHR4ZWYSW","short_pith_number":"pith:KTWN7CGZ","schema_version":"1.0","canonical_sha256":"54ecdf88d95a7bbba6ffb2cf1e66d895a075d9a3219ae9e4ac3b66bb742607d6","source":{"kind":"arxiv","id":"2504.13122","version":1},"attestation_state":"computed","paper":{"title":"VistaDPO: Video Hierarchical Spatial-Temporal Direct Preference Optimization for Large Video Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Haodong Chen, Hao Fei, Haojian Huang, Jinlan Fu, Meng Luo, Shengqiong Wu, Xinya Du","submitted_at":"2025-04-17T17:39:41Z","abstract_excerpt":"Large Video Models (LVMs) built upon Large Language Models (LLMs) have shown promise in video understanding but often suffer from misalignment with human intuition and video hallucination issues. To address these challenges, we introduce VistaDPO, a novel framework for Video Hierarchical Spatial-Temporal Direct Preference Optimization. VistaDPO enhances text-video preference alignment across three hierarchical levels: i) Instance Level, aligning overall video content with responses; ii) Temporal Level, aligning video temporal semantics with event descriptions; and iii) Perceptive Level, aligni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.13122","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-17T17:39:41Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"216671987c58f5142fbfba53e0c7b3c7fe5d31312a59caa38ae9abaef2edfe5b","abstract_canon_sha256":"c9d683c1d561444693df2cca09aa911a7be6bea9852d8c54ed00974021f785c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:37.770976Z","signature_b64":"vlr0zgHf4U/JoQNg3RyRaEA0eNThBWzBhOBGcxbSojTZEV+Suz7URUd/7frtfU/t/+CmItbHRuHN5+nQE27OBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54ecdf88d95a7bbba6ffb2cf1e66d895a075d9a3219ae9e4ac3b66bb742607d6","last_reissued_at":"2026-07-05T10:50:37.770270Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:37.770270Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VistaDPO: Video Hierarchical Spatial-Temporal Direct Preference Optimization for Large Video Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Hanwang Zhang, Haodong Chen, Hao Fei, Haojian Huang, Jinlan Fu, Meng Luo, Shengqiong Wu, Xinya Du","submitted_at":"2025-04-17T17:39:41Z","abstract_excerpt":"Large Video Models (LVMs) built upon Large Language Models (LLMs) have shown promise in video understanding but often suffer from misalignment with human intuition and video hallucination issues. To address these challenges, we introduce VistaDPO, a novel framework for Video Hierarchical Spatial-Temporal Direct Preference Optimization. VistaDPO enhances text-video preference alignment across three hierarchical levels: i) Instance Level, aligning overall video content with responses; ii) Temporal Level, aligning video temporal semantics with event descriptions; and iii) Perceptive Level, aligni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.13122","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.13122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.13122","created_at":"2026-07-05T10:50:37.770367+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.13122v1","created_at":"2026-07-05T10:50:37.770367+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.13122","created_at":"2026-07-05T10:50:37.770367+00:00"},{"alias_kind":"pith_short_12","alias_value":"KTWN7CGZLJ53","created_at":"2026-07-05T10:50:37.770367+00:00"},{"alias_kind":"pith_short_16","alias_value":"KTWN7CGZLJ53XJX7","created_at":"2026-07-05T10:50:37.770367+00:00"},{"alias_kind":"pith_short_8","alias_value":"KTWN7CGZ","created_at":"2026-07-05T10:50:37.770367+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11792","citing_title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18034","citing_title":"SignDPO: Multi-level Direct Preference Optimisation for Skeleton-based Gloss-free Sign Language Translation","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW","json":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW.json","graph_json":"https://pith.science/api/pith-number/KTWN7CGZLJ53XJX7WLHR4ZWYSW/graph.json","events_json":"https://pith.science/api/pith-number/KTWN7CGZLJ53XJX7WLHR4ZWYSW/events.json","paper":"https://pith.science/paper/KTWN7CGZ"},"agent_actions":{"view_html":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW","download_json":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW.json","view_paper":"https://pith.science/paper/KTWN7CGZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.13122&json=true","fetch_graph":"https://pith.science/api/pith-number/KTWN7CGZLJ53XJX7WLHR4ZWYSW/graph.json","fetch_events":"https://pith.science/api/pith-number/KTWN7CGZLJ53XJX7WLHR4ZWYSW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW/action/storage_attestation","attest_author":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW/action/author_attestation","sign_citation":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW/action/citation_signature","submit_replication":"https://pith.science/pith/KTWN7CGZLJ53XJX7WLHR4ZWYSW/action/replication_record"}},"created_at":"2026-07-05T10:50:37.770367+00:00","updated_at":"2026-07-05T10:50:37.770367+00:00"}