{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L4ZLRM2TWCQTVBV7Z254OZWCBV","short_pith_number":"pith:L4ZLRM2T","schema_version":"1.0","canonical_sha256":"5f32b8b353b0a13a86bfcebbc766c20d6343e70a0d240f68b4737d898947b7a1","source":{"kind":"arxiv","id":"2503.23905","version":2},"attestation_state":"computed","paper":{"title":"Boosting MLLM Reasoning with Text-Debiased Hint-GRPO","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Yao, Hao Jiang, Jie Song, Jingyuan Chen, Jinlong Liu, Mingli Song, Qihan Huang, Wanggui He, Weilong Dai","submitted_at":"2025-03-31T09:54:55Z","abstract_excerpt":"MLLM reasoning has drawn widespread research for its excellent problem-solving capability. Current reasoning methods fall into two types: PRM, which supervises the intermediate reasoning steps, and ORM, which supervises the final results. Recently, DeepSeek-R1 has challenged the traditional view that PRM outperforms ORM, which demonstrates strong generalization performance using an ORM method (i.e., GRPO). However, current MLLM's GRPO algorithms still struggle to handle challenging and complex multimodal reasoning tasks (e.g., mathematical reasoning). In this work, we reveal two problems that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23905","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-31T09:54:55Z","cross_cats_sorted":[],"title_canon_sha256":"20359b3bc997bdcfa2f488c5f1e09dc4f9be28b0e21fb3fc29c672bce7aa434b","abstract_canon_sha256":"8cf2a92968c91ccd977aa5dbed1072fc4b9cecc94458cc9dcd27ccfd033f06b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:15.461936Z","signature_b64":"XpatEuI/sY/MaV526m1jTiDUUPCeRBT93vDdEcq+iDmWIoqcamZLa1sJKfmokikXeSA0fstZ+VBv1SRSukHYCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f32b8b353b0a13a86bfcebbc766c20d6343e70a0d240f68b4737d898947b7a1","last_reissued_at":"2026-07-05T11:28:15.461486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:15.461486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Boosting MLLM Reasoning with Text-Debiased Hint-GRPO","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Yao, Hao Jiang, Jie Song, Jingyuan Chen, Jinlong Liu, Mingli Song, Qihan Huang, Wanggui He, Weilong Dai","submitted_at":"2025-03-31T09:54:55Z","abstract_excerpt":"MLLM reasoning has drawn widespread research for its excellent problem-solving capability. Current reasoning methods fall into two types: PRM, which supervises the intermediate reasoning steps, and ORM, which supervises the final results. Recently, DeepSeek-R1 has challenged the traditional view that PRM outperforms ORM, which demonstrates strong generalization performance using an ORM method (i.e., GRPO). However, current MLLM's GRPO algorithms still struggle to handle challenging and complex multimodal reasoning tasks (e.g., mathematical reasoning). In this work, we reveal two problems that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23905","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23905/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23905","created_at":"2026-07-05T11:28:15.461543+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23905v2","created_at":"2026-07-05T11:28:15.461543+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23905","created_at":"2026-07-05T11:28:15.461543+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4ZLRM2TWCQT","created_at":"2026-07-05T11:28:15.461543+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4ZLRM2TWCQTVBV7","created_at":"2026-07-05T11:28:15.461543+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4ZLRM2T","created_at":"2026-07-05T11:28:15.461543+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08572","citing_title":"Switch-Reasoner: Learn When to Think in Multitask Mixtures via Reinforcement Learning","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07674","citing_title":"Max Out GRPO Signal: Adaptive Trace Prefix Control for Hard Reasoning Problems","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2601.06794","citing_title":"No More Stale Feedback: Co-Evolving Critics for Open-World Agent Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03485","citing_title":"MHPR: Multidimensional Human Perception and Reasoning Benchmark for Large Vision-Languate Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01338","citing_title":"DiagramNet: An End-to-End Recognition Framework and Dataset for Non-Standard System-Level Diagrams","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV","json":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV.json","graph_json":"https://pith.science/api/pith-number/L4ZLRM2TWCQTVBV7Z254OZWCBV/graph.json","events_json":"https://pith.science/api/pith-number/L4ZLRM2TWCQTVBV7Z254OZWCBV/events.json","paper":"https://pith.science/paper/L4ZLRM2T"},"agent_actions":{"view_html":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV","download_json":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV.json","view_paper":"https://pith.science/paper/L4ZLRM2T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23905&json=true","fetch_graph":"https://pith.science/api/pith-number/L4ZLRM2TWCQTVBV7Z254OZWCBV/graph.json","fetch_events":"https://pith.science/api/pith-number/L4ZLRM2TWCQTVBV7Z254OZWCBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV/action/storage_attestation","attest_author":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV/action/author_attestation","sign_citation":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV/action/citation_signature","submit_replication":"https://pith.science/pith/L4ZLRM2TWCQTVBV7Z254OZWCBV/action/replication_record"}},"created_at":"2026-07-05T11:28:15.461543+00:00","updated_at":"2026-07-05T11:28:15.461543+00:00"}