{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JP7CKGEPDLUKIZESB7HFW5RUQI","short_pith_number":"pith:JP7CKGEP","schema_version":"1.0","canonical_sha256":"4bfe25188f1ae8a464920fce5b763482269dd1e37c5c864cd165c499e226ef88","source":{"kind":"arxiv","id":"2505.16673","version":1},"attestation_state":"computed","paper":{"title":"R1-ShareVL: Incentivizing Reasoning Capability of Multimodal Large Language Models via Share-GRPO","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Fei Su, Huanjin Yao, Jiaxing Huang, Jingyi Zhang, Li Shen, Minghui Qiu, Min Yang, Qixiang Yin, Wenhao Wu, Yibo Wang","submitted_at":"2025-05-22T13:39:32Z","abstract_excerpt":"In this work, we aim to incentivize the reasoning ability of Multimodal Large Language Models (MLLMs) via reinforcement learning (RL) and develop an effective approach that mitigates the sparse reward and advantage vanishing issues during RL. To this end, we propose Share-GRPO, a novel RL approach that tackle these issues by exploring and sharing diverse reasoning trajectories over expanded question space. Specifically, Share-GRPO first expands the question space for a given question via data transformation techniques, and then encourages MLLM to effectively explore diverse reasoning trajector"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16673","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-22T13:39:32Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"39a4373a007f0be7594afe90728ca295c0a5720be3df651197cde877567db605","abstract_canon_sha256":"145d7f3fa8f2574c5c30d50b8650cdce0df5ddab8843395592b0366f4f9adb75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:38.877276Z","signature_b64":"jphVygc9dPS59yRfO986pPItpAT8YLjMFuTVkaLz7cDj2D2+A3RmgReNgTGb/OdOjRiwBs2ib2eJuA/flzSyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4bfe25188f1ae8a464920fce5b763482269dd1e37c5c864cd165c499e226ef88","last_reissued_at":"2026-07-05T11:07:38.876192Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:38.876192Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"R1-ShareVL: Incentivizing Reasoning Capability of Multimodal Large Language Models via Share-GRPO","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Fei Su, Huanjin Yao, Jiaxing Huang, Jingyi Zhang, Li Shen, Minghui Qiu, Min Yang, Qixiang Yin, Wenhao Wu, Yibo Wang","submitted_at":"2025-05-22T13:39:32Z","abstract_excerpt":"In this work, we aim to incentivize the reasoning ability of Multimodal Large Language Models (MLLMs) via reinforcement learning (RL) and develop an effective approach that mitigates the sparse reward and advantage vanishing issues during RL. To this end, we propose Share-GRPO, a novel RL approach that tackle these issues by exploring and sharing diverse reasoning trajectories over expanded question space. Specifically, Share-GRPO first expands the question space for a given question via data transformation techniques, and then encourages MLLM to effectively explore diverse reasoning trajector"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16673","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16673","created_at":"2026-07-05T11:07:38.876256+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16673v1","created_at":"2026-07-05T11:07:38.876256+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16673","created_at":"2026-07-05T11:07:38.876256+00:00"},{"alias_kind":"pith_short_12","alias_value":"JP7CKGEPDLUK","created_at":"2026-07-05T11:07:38.876256+00:00"},{"alias_kind":"pith_short_16","alias_value":"JP7CKGEPDLUKIZES","created_at":"2026-07-05T11:07:38.876256+00:00"},{"alias_kind":"pith_short_8","alias_value":"JP7CKGEP","created_at":"2026-07-05T11:07:38.876256+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08572","citing_title":"Switch-Reasoner: Learn When to Think in Multitask Mixtures via Reinforcement Learning","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01707","citing_title":"LASER: A Corrective Lens for LVLMs via Visual Attention Preservation and Sink Suppression","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29812","citing_title":"Consistency as Inductive Bias: Learning Cross-View Invariance for Robust Multimodal Reasoning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19852","citing_title":"Are Tools Always Beneficial? Learning to Invoke Tools Adaptively for Dual-Mode Multimodal LLM Reasoning","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12937","citing_title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01824","citing_title":"STRIVE: Structured Spatiotemporal Exploration for Reinforcement Learning in Video Question Answering","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27083","citing_title":"Co-Evolving Policy Distillation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09614","citing_title":"Reflection Anchors for Propagation-Aware Visual Retention in Long-Chain Multimodal Reasoning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09262","citing_title":"Reinforcing Multimodal Reasoning Against Visual Degradation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01882","citing_title":"Chart-FR1: Visual Focus-Driven Fine-Grained Reasoning on Dense Charts","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI","json":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI.json","graph_json":"https://pith.science/api/pith-number/JP7CKGEPDLUKIZESB7HFW5RUQI/graph.json","events_json":"https://pith.science/api/pith-number/JP7CKGEPDLUKIZESB7HFW5RUQI/events.json","paper":"https://pith.science/paper/JP7CKGEP"},"agent_actions":{"view_html":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI","download_json":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI.json","view_paper":"https://pith.science/paper/JP7CKGEP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16673&json=true","fetch_graph":"https://pith.science/api/pith-number/JP7CKGEPDLUKIZESB7HFW5RUQI/graph.json","fetch_events":"https://pith.science/api/pith-number/JP7CKGEPDLUKIZESB7HFW5RUQI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI/action/storage_attestation","attest_author":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI/action/author_attestation","sign_citation":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI/action/citation_signature","submit_replication":"https://pith.science/pith/JP7CKGEPDLUKIZESB7HFW5RUQI/action/replication_record"}},"created_at":"2026-07-05T11:07:38.876256+00:00","updated_at":"2026-07-05T11:07:38.876256+00:00"}