{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:H7VCLRCOLN74KHEXYEPNSLLSAT","short_pith_number":"pith:H7VCLRCO","schema_version":"1.0","canonical_sha256":"3fea25c44e5b7fc51c97c11ed92d7204eab6861e568ef7e9a60905b0b2ac6005","source":{"kind":"arxiv","id":"2505.22334","version":2},"attestation_state":"computed","paper":{"title":"Advancing Multimodal Reasoning via Reinforcement Learning with Cold Start","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chen Wang, Kaipeng Zheng, Lai Wei, Lichao Sun, Linghe Kong, Weiran Huang, Yue Wang, Yuting Li","submitted_at":"2025-05-28T13:21:38Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have demonstrated impressive chain-of-thought reasoning capabilities, with reinforcement learning (RL) playing a crucial role in this progress. While \"aha moment\" patterns--where models exhibit self-correction through reflection--are often attributed to emergent properties from RL, we first demonstrate that these patterns exist in multimodal LLMs (MLLMs) prior to RL training but may not necessarily correlate with improved reasoning performance. Building on these insights, we present a comprehensive study on enhancing multimodal reasoning thro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22334","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T13:21:38Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"ee85526c236f909ed56918d4b82f4bb650e5c2952eff3d2dd61bb4e4a04aad72","abstract_canon_sha256":"5bbf77f2485727b7237d9b0862ad66f6c1a6777d1c1eac2383b6173889a01a4c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:42.310227Z","signature_b64":"Co3r8a1E36Ya7Ig5EgZULYjKUfVA7DytVNoaqvq/OgX3NPRThqB3cee1OUXJ12RwhwGSgSEE5KmtX0i8MlOtCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3fea25c44e5b7fc51c97c11ed92d7204eab6861e568ef7e9a60905b0b2ac6005","last_reissued_at":"2026-07-05T11:41:42.309630Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:42.309630Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Advancing Multimodal Reasoning via Reinforcement Learning with Cold Start","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chen Wang, Kaipeng Zheng, Lai Wei, Lichao Sun, Linghe Kong, Weiran Huang, Yue Wang, Yuting Li","submitted_at":"2025-05-28T13:21:38Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have demonstrated impressive chain-of-thought reasoning capabilities, with reinforcement learning (RL) playing a crucial role in this progress. While \"aha moment\" patterns--where models exhibit self-correction through reflection--are often attributed to emergent properties from RL, we first demonstrate that these patterns exist in multimodal LLMs (MLLMs) prior to RL training but may not necessarily correlate with improved reasoning performance. Building on these insights, we present a comprehensive study on enhancing multimodal reasoning thro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22334","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22334/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22334","created_at":"2026-07-05T11:41:42.309693+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22334v2","created_at":"2026-07-05T11:41:42.309693+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22334","created_at":"2026-07-05T11:41:42.309693+00:00"},{"alias_kind":"pith_short_12","alias_value":"H7VCLRCOLN74","created_at":"2026-07-05T11:41:42.309693+00:00"},{"alias_kind":"pith_short_16","alias_value":"H7VCLRCOLN74KHEX","created_at":"2026-07-05T11:41:42.309693+00:00"},{"alias_kind":"pith_short_8","alias_value":"H7VCLRCO","created_at":"2026-07-05T11:41:42.309693+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24375","citing_title":"Distilling Game Code World Model Generation into Lightweight Large Language Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29984","citing_title":"Be Faithful When Response: Returning Fluent and Grounded Answers for Vision-Language Models Reinforcement Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12163","citing_title":"Self-Consistent Latent Reasoning: Long Latent Sequence Reasoning for Vision-Language Model","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12163","citing_title":"Self-Consistent Latent Reasoning: Long Latent Sequence Reasoning for Vision-Language Model","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16896","citing_title":"ProtoCycle: Reflective Tool-Augmented Planning for Text-Guided Protein Design","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT","json":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT.json","graph_json":"https://pith.science/api/pith-number/H7VCLRCOLN74KHEXYEPNSLLSAT/graph.json","events_json":"https://pith.science/api/pith-number/H7VCLRCOLN74KHEXYEPNSLLSAT/events.json","paper":"https://pith.science/paper/H7VCLRCO"},"agent_actions":{"view_html":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT","download_json":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT.json","view_paper":"https://pith.science/paper/H7VCLRCO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22334&json=true","fetch_graph":"https://pith.science/api/pith-number/H7VCLRCOLN74KHEXYEPNSLLSAT/graph.json","fetch_events":"https://pith.science/api/pith-number/H7VCLRCOLN74KHEXYEPNSLLSAT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT/action/storage_attestation","attest_author":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT/action/author_attestation","sign_citation":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT/action/citation_signature","submit_replication":"https://pith.science/pith/H7VCLRCOLN74KHEXYEPNSLLSAT/action/replication_record"}},"created_at":"2026-07-05T11:41:42.309693+00:00","updated_at":"2026-07-05T11:41:42.309693+00:00"}