{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:JP7CKGEPDLUKIZESB7HFW5RUQI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"145d7f3fa8f2574c5c30d50b8650cdce0df5ddab8843395592b0366f4f9adb75","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-22T13:39:32Z","title_canon_sha256":"39a4373a007f0be7594afe90728ca295c0a5720be3df651197cde877567db605"},"schema_version":"1.0","source":{"id":"2505.16673","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.16673","created_at":"2026-07-05T11:07:38Z"},{"alias_kind":"arxiv_version","alias_value":"2505.16673v1","created_at":"2026-07-05T11:07:38Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16673","created_at":"2026-07-05T11:07:38Z"},{"alias_kind":"pith_short_12","alias_value":"JP7CKGEPDLUK","created_at":"2026-07-05T11:07:38Z"},{"alias_kind":"pith_short_16","alias_value":"JP7CKGEPDLUKIZES","created_at":"2026-07-05T11:07:38Z"},{"alias_kind":"pith_short_8","alias_value":"JP7CKGEP","created_at":"2026-07-05T11:07:38Z"}],"graph_snapshots":[{"event_id":"sha256:b14a3b446712304cd2a2ff024140e5217cca3c69121bc882d67694a900ee58f8","target":"graph","created_at":"2026-07-05T11:07:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.16673/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this work, we aim to incentivize the reasoning ability of Multimodal Large Language Models (MLLMs) via reinforcement learning (RL) and develop an effective approach that mitigates the sparse reward and advantage vanishing issues during RL. To this end, we propose Share-GRPO, a novel RL approach that tackle these issues by exploring and sharing diverse reasoning trajectories over expanded question space. Specifically, Share-GRPO first expands the question space for a given question via data transformation techniques, and then encourages MLLM to effectively explore diverse reasoning trajector","authors_text":"Dacheng Tao, Fei Su, Huanjin Yao, Jiaxing Huang, Jingyi Zhang, Li Shen, Minghui Qiu, Min Yang, Qixiang Yin, Wenhao Wu, Yibo Wang","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-22T13:39:32Z","title":"R1-ShareVL: Incentivizing Reasoning Capability of Multimodal Large Language Models via Share-GRPO"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16673","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:330c1511273bf45a8c864e86a2000bd808f6bcf75a297ebe1c7d2c8d194d4f58","target":"record","created_at":"2026-07-05T11:07:38Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"145d7f3fa8f2574c5c30d50b8650cdce0df5ddab8843395592b0366f4f9adb75","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-22T13:39:32Z","title_canon_sha256":"39a4373a007f0be7594afe90728ca295c0a5720be3df651197cde877567db605"},"schema_version":"1.0","source":{"id":"2505.16673","kind":"arxiv","version":1}},"canonical_sha256":"4bfe25188f1ae8a464920fce5b763482269dd1e37c5c864cd165c499e226ef88","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4bfe25188f1ae8a464920fce5b763482269dd1e37c5c864cd165c499e226ef88","first_computed_at":"2026-07-05T11:07:38.876192Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:07:38.876192Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jphVygc9dPS59yRfO986pPItpAT8YLjMFuTVkaLz7cDj2D2+A3RmgReNgTGb/OdOjRiwBs2ib2eJuA/flzSyDA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:07:38.877276Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.16673","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:330c1511273bf45a8c864e86a2000bd808f6bcf75a297ebe1c7d2c8d194d4f58","sha256:b14a3b446712304cd2a2ff024140e5217cca3c69121bc882d67694a900ee58f8"],"state_sha256":"2ae8618d03786544ac0c775324ab62bce3c4dc3990830d890fbeb96538a73553"}