{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:M2OCSINRGJHP3L5DCBHD532RXD","short_pith_number":"pith:M2OCSINR","schema_version":"1.0","canonical_sha256":"669c2921b1324efdafa3104e3eef51b8c67a052d6bca2f79ba251343d1a67554","source":{"kind":"arxiv","id":"2505.02835","version":2},"attestation_state":"computed","paper":{"title":"R1-Reward: Training Multimodal Reward Model Through Stable Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Wen, Changyi Liu, Chaoyou Fu, Fan Yang, Haojie Ding, Jiankang Chen, Kaibing Chen, Kaiyu Jiang, Kaiyu Tang, Liang Wang, Tianke Zhang, Tingting Gao, Xiao Hu, Xingyu Lu, Yi-Fan Zhang, Zhang Zhang","submitted_at":"2025-05-05T17:59:50Z","abstract_excerpt":"Multimodal Reward Models (MRMs) play a crucial role in enhancing the performance of Multimodal Large Language Models (MLLMs). While recent advancements have primarily focused on improving the model structure and training data of MRMs, there has been limited exploration into the effectiveness of long-term reasoning capabilities for reward modeling and how to activate these capabilities in MRMs. In this paper, we explore how Reinforcement Learning (RL) can be used to improve reward modeling. Specifically, we reformulate the reward modeling problem as a rule-based RL task. However, we observe tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.02835","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-05T17:59:50Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6c6970d82438f5cd46304a0fdf4b230fba0157a2c418a0e6e8261033894a6a89","abstract_canon_sha256":"3ccd0639fd4778665b2c31c6c0f050e25e9d4c0487406e944a09f8a3b5db6eb4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:44.787995Z","signature_b64":"xlF2xEIAYepFkgoSLf//xpX9/M4S4CuOKGo8yc/XhFL2405It7ylM90U4jA/YpL93g6RPPk7QJ/lkO+RPw4ZBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"669c2921b1324efdafa3104e3eef51b8c67a052d6bca2f79ba251343d1a67554","last_reissued_at":"2026-07-05T11:00:44.787402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:44.787402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"R1-Reward: Training Multimodal Reward Model Through Stable Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Wen, Changyi Liu, Chaoyou Fu, Fan Yang, Haojie Ding, Jiankang Chen, Kaibing Chen, Kaiyu Jiang, Kaiyu Tang, Liang Wang, Tianke Zhang, Tingting Gao, Xiao Hu, Xingyu Lu, Yi-Fan Zhang, Zhang Zhang","submitted_at":"2025-05-05T17:59:50Z","abstract_excerpt":"Multimodal Reward Models (MRMs) play a crucial role in enhancing the performance of Multimodal Large Language Models (MLLMs). While recent advancements have primarily focused on improving the model structure and training data of MRMs, there has been limited exploration into the effectiveness of long-term reasoning capabilities for reward modeling and how to activate these capabilities in MRMs. In this paper, we explore how Reinforcement Learning (RL) can be used to improve reward modeling. Specifically, we reformulate the reward modeling problem as a rule-based RL task. However, we observe tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.02835","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.02835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.02835","created_at":"2026-07-05T11:00:44.787476+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.02835v2","created_at":"2026-07-05T11:00:44.787476+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.02835","created_at":"2026-07-05T11:00:44.787476+00:00"},{"alias_kind":"pith_short_12","alias_value":"M2OCSINRGJHP","created_at":"2026-07-05T11:00:44.787476+00:00"},{"alias_kind":"pith_short_16","alias_value":"M2OCSINRGJHP3L5D","created_at":"2026-07-05T11:00:44.787476+00:00"},{"alias_kind":"pith_short_8","alias_value":"M2OCSINR","created_at":"2026-07-05T11:00:44.787476+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09064","citing_title":"See More, Think Deeper: Query-Expanded Visual Evidence and Answer-Clue Guided Reflection for Long Video Understanding","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25437","citing_title":"Does Seeing More Mean Knowing More? Mono-Anchored Advantage Normalization for Multi-Source Visual Reasoning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22104","citing_title":"OPERA: An Agent for Image Restoration with End-to-End Joint Planning-Execution Optimization","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2505.24499","citing_title":"Reason-SVG: Enhancing Structured Reasoning for Vector Graphics Generation with Reinforcement Learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05922","citing_title":"Think, then Score: Decoupled Reasoning and Scoring for Video Reward Modeling","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09422","citing_title":"Perception Without Engagement: Dissecting the Causal Discovery Deficit in LMMs","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09269","citing_title":"DeltaRubric: Generative Multimodal Reward Modeling via Joint Planning and Verification","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24339","citing_title":"See Further, Think Deeper: Advancing VLM's Reasoning Ability with Low-level Visual Cues and Reflection","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05965","citing_title":"Beyond Uniform Credit Assignment: Selective Eligibility Traces for RLVR","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19544","citing_title":"DT2IT-MRM: Debiased Preference Construction and Iterative Training for Multimodal Reward Modeling","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08905","citing_title":"StaRPO: Stability-Augmented Reinforcement Policy Optimization","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07872","citing_title":"Video Understanding Reward Modeling: A Robust Benchmark and Performant Reward Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14910","citing_title":"Reward-Aware Trajectory Shaping for Few-step Visual Generation","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD","json":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD.json","graph_json":"https://pith.science/api/pith-number/M2OCSINRGJHP3L5DCBHD532RXD/graph.json","events_json":"https://pith.science/api/pith-number/M2OCSINRGJHP3L5DCBHD532RXD/events.json","paper":"https://pith.science/paper/M2OCSINR"},"agent_actions":{"view_html":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD","download_json":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD.json","view_paper":"https://pith.science/paper/M2OCSINR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.02835&json=true","fetch_graph":"https://pith.science/api/pith-number/M2OCSINRGJHP3L5DCBHD532RXD/graph.json","fetch_events":"https://pith.science/api/pith-number/M2OCSINRGJHP3L5DCBHD532RXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD/action/storage_attestation","attest_author":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD/action/author_attestation","sign_citation":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD/action/citation_signature","submit_replication":"https://pith.science/pith/M2OCSINRGJHP3L5DCBHD532RXD/action/replication_record"}},"created_at":"2026-07-05T11:00:44.787476+00:00","updated_at":"2026-07-05T11:00:44.787476+00:00"}