{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CA6O5YWQN7YWHXGDTEIMY74T3V","short_pith_number":"pith:CA6O5YWQ","schema_version":"1.0","canonical_sha256":"103ceee2d06ff163dcc39910cc7f93dd750e3b18e926572582904db2422138af","source":{"kind":"arxiv","id":"2509.00484","version":1},"attestation_state":"computed","paper":{"title":"VideoRewardBench: Comprehensive Evaluation of Multimodal Reward Models for Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jiansheng Wei, Jin Xu, Xiaojian Huang, Xinzhi Wang, Xuejin Chen, Zhihong Zhang, Zhuodong Luo","submitted_at":"2025-08-30T12:50:55Z","abstract_excerpt":"Multimodal reward models (MRMs) play a crucial role in the training, inference, and evaluation of Large Vision Language Models (LVLMs) by assessing response quality. However, existing benchmarks for evaluating MRMs in the video domain suffer from a limited number and diversity of questions, a lack of comprehensive evaluation dimensions, and inadequate evaluation of diverse types of MRMs. To address these gaps, we introduce VideoRewardBench, the first comprehensive benchmark covering four core aspects of video understanding: perception, knowledge, reasoning, and safety. Through our AI-assisted "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00484","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-30T12:50:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"79bd7d2d297d5efa75c06b7d497fe2fd27247487bfe651f99de1aa0156d97c91","abstract_canon_sha256":"44aa4d54ba3040c0d22c8bf831fc06a0fadb91e53537796e66e2ec5197e5054d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:17.058610Z","signature_b64":"r5OvFn/xSIueUY1Ukqb6PYwKHNXUL1KWDxXeRWQSNjt1E0LI0G8MK9q7dfjWOguL+m65OnCgsMZIaq/1KE3DCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"103ceee2d06ff163dcc39910cc7f93dd750e3b18e926572582904db2422138af","last_reissued_at":"2026-07-05T12:02:17.057952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:17.057952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoRewardBench: Comprehensive Evaluation of Multimodal Reward Models for Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jiansheng Wei, Jin Xu, Xiaojian Huang, Xinzhi Wang, Xuejin Chen, Zhihong Zhang, Zhuodong Luo","submitted_at":"2025-08-30T12:50:55Z","abstract_excerpt":"Multimodal reward models (MRMs) play a crucial role in the training, inference, and evaluation of Large Vision Language Models (LVLMs) by assessing response quality. However, existing benchmarks for evaluating MRMs in the video domain suffer from a limited number and diversity of questions, a lack of comprehensive evaluation dimensions, and inadequate evaluation of diverse types of MRMs. To address these gaps, we introduce VideoRewardBench, the first comprehensive benchmark covering four core aspects of video understanding: perception, knowledge, reasoning, and safety. Through our AI-assisted "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00484","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00484/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00484","created_at":"2026-07-05T12:02:17.058054+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00484v1","created_at":"2026-07-05T12:02:17.058054+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00484","created_at":"2026-07-05T12:02:17.058054+00:00"},{"alias_kind":"pith_short_12","alias_value":"CA6O5YWQN7YW","created_at":"2026-07-05T12:02:17.058054+00:00"},{"alias_kind":"pith_short_16","alias_value":"CA6O5YWQN7YWHXGD","created_at":"2026-07-05T12:02:17.058054+00:00"},{"alias_kind":"pith_short_8","alias_value":"CA6O5YWQ","created_at":"2026-07-05T12:02:17.058054+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07872","citing_title":"Video Understanding Reward Modeling: A Robust Benchmark and Performant Reward Models","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V","json":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V.json","graph_json":"https://pith.science/api/pith-number/CA6O5YWQN7YWHXGDTEIMY74T3V/graph.json","events_json":"https://pith.science/api/pith-number/CA6O5YWQN7YWHXGDTEIMY74T3V/events.json","paper":"https://pith.science/paper/CA6O5YWQ"},"agent_actions":{"view_html":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V","download_json":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V.json","view_paper":"https://pith.science/paper/CA6O5YWQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00484&json=true","fetch_graph":"https://pith.science/api/pith-number/CA6O5YWQN7YWHXGDTEIMY74T3V/graph.json","fetch_events":"https://pith.science/api/pith-number/CA6O5YWQN7YWHXGDTEIMY74T3V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V/action/storage_attestation","attest_author":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V/action/author_attestation","sign_citation":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V/action/citation_signature","submit_replication":"https://pith.science/pith/CA6O5YWQN7YWHXGDTEIMY74T3V/action/replication_record"}},"created_at":"2026-07-05T12:02:17.058054+00:00","updated_at":"2026-07-05T12:02:17.058054+00:00"}