{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YV2S2RHPRUL2E54JHXQK47VCNH","short_pith_number":"pith:YV2S2RHP","schema_version":"1.0","canonical_sha256":"c5752d44ef8d17a277893de0ae7ea269d5531b569c363317af967565b63eec5d","source":{"kind":"arxiv","id":"2503.10291","version":1},"attestation_state":"computed","paper":{"title":"VisualPRM: An Effective Process Reward Model for Multimodal Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Haodong Duan, Jifeng Dai, Jinguo Zhu, Lewei Lu, Lianjie Chen, Shenglong Ye, Weiyun Wang, Wenhai Wang, Xiangyu Zhao, Xizhou Zhu, Yangzhou Liu, Yue Cao, Yu Qiao, Zhangwei Gao, Zhe Chen","submitted_at":"2025-03-13T12:03:37Z","abstract_excerpt":"We introduce VisualPRM, an advanced multimodal Process Reward Model (PRM) with 8B parameters, which improves the reasoning abilities of existing Multimodal Large Language Models (MLLMs) across different model scales and families with Best-of-N (BoN) evaluation strategies. Specifically, our model improves the reasoning performance of three types of MLLMs and four different model scales. Even when applied to the highly capable InternVL2.5-78B, it achieves a 5.9-point improvement across seven multimodal reasoning benchmarks. Experimental results show that our model exhibits superior performance c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10291","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T12:03:37Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2520e0577d4dd68b6676089965df733e9e4ca941a14e298da024840adae489c5","abstract_canon_sha256":"b0722d520b6034a4443ce9c2a0c881e71e31c908dac1435fbe54e2c1113b308a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:42.374357Z","signature_b64":"4lrPY2fAY5Lye7zp1X/PSyo8xbMRWAXzMrtznT/Ig+7cUKOBpvh27Pkvfm93ZWZivQ0FhB/aorilUIFiJDbsCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5752d44ef8d17a277893de0ae7ea269d5531b569c363317af967565b63eec5d","last_reissued_at":"2026-07-05T10:30:42.373834Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:42.373834Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisualPRM: An Effective Process Reward Model for Multimodal Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Haodong Duan, Jifeng Dai, Jinguo Zhu, Lewei Lu, Lianjie Chen, Shenglong Ye, Weiyun Wang, Wenhai Wang, Xiangyu Zhao, Xizhou Zhu, Yangzhou Liu, Yue Cao, Yu Qiao, Zhangwei Gao, Zhe Chen","submitted_at":"2025-03-13T12:03:37Z","abstract_excerpt":"We introduce VisualPRM, an advanced multimodal Process Reward Model (PRM) with 8B parameters, which improves the reasoning abilities of existing Multimodal Large Language Models (MLLMs) across different model scales and families with Best-of-N (BoN) evaluation strategies. Specifically, our model improves the reasoning performance of three types of MLLMs and four different model scales. Even when applied to the highly capable InternVL2.5-78B, it achieves a 5.9-point improvement across seven multimodal reasoning benchmarks. Experimental results show that our model exhibits superior performance c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10291","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10291","created_at":"2026-07-05T10:30:42.373893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10291v1","created_at":"2026-07-05T10:30:42.373893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10291","created_at":"2026-07-05T10:30:42.373893+00:00"},{"alias_kind":"pith_short_12","alias_value":"YV2S2RHPRUL2","created_at":"2026-07-05T10:30:42.373893+00:00"},{"alias_kind":"pith_short_16","alias_value":"YV2S2RHPRUL2E54J","created_at":"2026-07-05T10:30:42.373893+00:00"},{"alias_kind":"pith_short_8","alias_value":"YV2S2RHP","created_at":"2026-07-05T10:30:42.373893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18441","citing_title":"Reasoning as Intersection: Consensus-Frame Alignment for Visual Focus in Video-MLLMs","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07801","citing_title":"Improving Multimodal Reasoning via Worst Dimension Optimization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04579","citing_title":"SCI-PRM: A Tool Aware Process Reward Model for Scientific Reasoning Verification","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01667","citing_title":"ATLAS: Agentic Test-time Learning-to-Allocate Scaling","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29119","citing_title":"PRO-CUA: Process-Reward Optimization for Computer Use Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00148","citing_title":"StemBind: When MLLMs Get Lost Between Rules and Instances in Abstract Visual Reasoning","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04300","citing_title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2508.03556","citing_title":"VRPRM: Process Reward Modeling via Visual Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17877","citing_title":"PAIR: Prefix-Aware Internal Reward Model for Multi-Turn Agent Optimization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19538","citing_title":"CaptchaMind: Training CAPTCHA Solvers via Reinforcement Learning with Explicit Reasoning Supervision","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17352","citing_title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14186","citing_title":"LLMs Know When They Know, but Do Not Act on It: A Metacognitive Harness for Test-time Scaling","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12163","citing_title":"Self-Consistent Latent Reasoning: Long Latent Sequence Reasoning for Vision-Language Model","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13467","citing_title":"PDCR: Perception-Decomposed Confidence Reward for Vision-Language Reasoning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12163","citing_title":"Self-Consistent Latent Reasoning: Long Latent Sequence Reasoning for Vision-Language Model","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10850","citing_title":"Verification Mirage: Mapping the Reliability Boundary of Self-Verification in Medical VQA","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20755","citing_title":"V-tableR1: Process-Supervised Multimodal Table Reasoning with Critic-Guided Policy Optimization","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19544","citing_title":"DT2IT-MRM: Debiased Preference Construction and Iterative Training for Multimodal Reward Modeling","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01882","citing_title":"Chart-FR1: Visual Focus-Driven Fine-Grained Reasoning on Dense Charts","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":126,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH","json":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH.json","graph_json":"https://pith.science/api/pith-number/YV2S2RHPRUL2E54JHXQK47VCNH/graph.json","events_json":"https://pith.science/api/pith-number/YV2S2RHPRUL2E54JHXQK47VCNH/events.json","paper":"https://pith.science/paper/YV2S2RHP"},"agent_actions":{"view_html":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH","download_json":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH.json","view_paper":"https://pith.science/paper/YV2S2RHP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10291&json=true","fetch_graph":"https://pith.science/api/pith-number/YV2S2RHPRUL2E54JHXQK47VCNH/graph.json","fetch_events":"https://pith.science/api/pith-number/YV2S2RHPRUL2E54JHXQK47VCNH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH/action/storage_attestation","attest_author":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH/action/author_attestation","sign_citation":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH/action/citation_signature","submit_replication":"https://pith.science/pith/YV2S2RHPRUL2E54JHXQK47VCNH/action/replication_record"}},"created_at":"2026-07-05T10:30:42.373893+00:00","updated_at":"2026-07-05T10:30:42.373893+00:00"}