{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:75K2LPHQORUZOZR473KDNN3IRV","short_pith_number":"pith:75K2LPHQ","schema_version":"1.0","canonical_sha256":"ff55a5bcf0746997663cfed436b7688d4c524c88df7bb9ecabd2a852ad4595bf","source":{"kind":"arxiv","id":"2502.14191","version":1},"attestation_state":"computed","paper":{"title":"Multimodal RewardBench: Holistic Evaluation of Reward Models for Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Luke Zettlemoyer, Marjan Ghazvininejad, Michihiro Yasunaga","submitted_at":"2025-02-20T01:48:13Z","abstract_excerpt":"Reward models play an essential role in training vision-language models (VLMs) by assessing output quality to enable aligning with human preferences. Despite their importance, the research community lacks comprehensive open benchmarks for evaluating multimodal reward models in VLMs. To address this gap, we introduce Multimodal RewardBench, an expert-annotated benchmark covering six domains: general correctness, preference, knowledge, reasoning, safety, and visual question-answering. Our dataset comprises 5,211 annotated (prompt, chosen response, rejected response) triplets collected from vario"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14191","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-20T01:48:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5cb54b92895b3a033dd568edbeff64fe13959adff57f28f4013b951e2ba2df80","abstract_canon_sha256":"d3597e6849a81489f1ebc9aae164abc142e07dae1d4a94927d51a8ee1f721f3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:22.464800Z","signature_b64":"bUV/3MRt6XDbKXZgb+BTjIqmM+C53+0N3Zc9q/rGdIBct2W2OKIulItSivtkl4AR3OGPPMDuaHl5C54zUzPGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff55a5bcf0746997663cfed436b7688d4c524c88df7bb9ecabd2a852ad4595bf","last_reissued_at":"2026-07-05T10:17:22.464177Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:22.464177Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multimodal RewardBench: Holistic Evaluation of Reward Models for Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Luke Zettlemoyer, Marjan Ghazvininejad, Michihiro Yasunaga","submitted_at":"2025-02-20T01:48:13Z","abstract_excerpt":"Reward models play an essential role in training vision-language models (VLMs) by assessing output quality to enable aligning with human preferences. Despite their importance, the research community lacks comprehensive open benchmarks for evaluating multimodal reward models in VLMs. To address this gap, we introduce Multimodal RewardBench, an expert-annotated benchmark covering six domains: general correctness, preference, knowledge, reasoning, safety, and visual question-answering. Our dataset comprises 5,211 annotated (prompt, chosen response, rejected response) triplets collected from vario"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14191","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14191/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14191","created_at":"2026-07-05T10:17:22.464243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14191v1","created_at":"2026-07-05T10:17:22.464243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14191","created_at":"2026-07-05T10:17:22.464243+00:00"},{"alias_kind":"pith_short_12","alias_value":"75K2LPHQORUZ","created_at":"2026-07-05T10:17:22.464243+00:00"},{"alias_kind":"pith_short_16","alias_value":"75K2LPHQORUZOZR4","created_at":"2026-07-05T10:17:22.464243+00:00"},{"alias_kind":"pith_short_8","alias_value":"75K2LPHQ","created_at":"2026-07-05T10:17:22.464243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.32034","citing_title":"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09269","citing_title":"DeltaRubric: Generative Multimodal Reward Modeling via Joint Planning and Verification","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21718","citing_title":"Building a Precise Video Language with Human-AI Oversight","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19544","citing_title":"DT2IT-MRM: Debiased Preference Construction and Iterative Training for Multimodal Reward Modeling","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19405","citing_title":"Lost in Translation: Do LVLM Judges Generalize Across Languages?","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13029","citing_title":"Visual Preference Optimization with Rubric Rewards","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07872","citing_title":"Video Understanding Reward Modeling: A Robust Benchmark and Performant Reward Models","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV","json":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV.json","graph_json":"https://pith.science/api/pith-number/75K2LPHQORUZOZR473KDNN3IRV/graph.json","events_json":"https://pith.science/api/pith-number/75K2LPHQORUZOZR473KDNN3IRV/events.json","paper":"https://pith.science/paper/75K2LPHQ"},"agent_actions":{"view_html":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV","download_json":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV.json","view_paper":"https://pith.science/paper/75K2LPHQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14191&json=true","fetch_graph":"https://pith.science/api/pith-number/75K2LPHQORUZOZR473KDNN3IRV/graph.json","fetch_events":"https://pith.science/api/pith-number/75K2LPHQORUZOZR473KDNN3IRV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV/action/storage_attestation","attest_author":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV/action/author_attestation","sign_citation":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV/action/citation_signature","submit_replication":"https://pith.science/pith/75K2LPHQORUZOZR473KDNN3IRV/action/replication_record"}},"created_at":"2026-07-05T10:17:22.464243+00:00","updated_at":"2026-07-05T10:17:22.464243+00:00"}