{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P6YLOMPZUJQGG436AM7DNFHHV2","short_pith_number":"pith:P6YLOMPZ","schema_version":"1.0","canonical_sha256":"7fb0b731f9a26063737e033e3694e7ae979714146177762eb9ae26ee22700b02","source":{"kind":"arxiv","id":"2412.18319","version":2},"attestation_state":"computed","paper":{"title":"Mulberry: Empowering MLLM with o1-like Reasoning and Reflection via Collective Monte Carlo Tree Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Haocheng Feng, Huanjin Yao, Jiaxing Huang, Jingyi Zhang, Li Shen, Shunyu Liu, Wenhao Wu, Yibo Wang, Yingjie Wang, Yuxin Song","submitted_at":"2024-12-24T10:07:51Z","abstract_excerpt":"In this work, we aim to develop an MLLM that understands and solves questions by learning to create each intermediate step of the reasoning involved till the final answer. To this end, we propose Collective Monte Carlo Tree Search (CoMCTS), a new learning-to-reason method for MLLMs, which introduces the concept of collective learning into ``tree search'' for effective and efficient reasoning-path searching and learning. The core idea of CoMCTS is to leverage collective knowledge from multiple models to collaboratively conjecture, search and identify effective reasoning paths toward correct ans"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18319","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-24T10:07:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"41c96ad1044493da6b945f2ac7d256eb277a8ccfa0e85a93472018843c1335db","abstract_canon_sha256":"6efb233d453797d912a1a911bdce3a83421b1868b1de2e34a98d16bd436dd37a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:36.075950Z","signature_b64":"7/gpj7wB+IenjMBoIP5i7Rm5rzDRqCM40MJdKO1rlMYZbTJ2KBB4MwvsVL4kWJNLKkorsO3pdUCLG6LtqTiRDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fb0b731f9a26063737e033e3694e7ae979714146177762eb9ae26ee22700b02","last_reissued_at":"2026-07-05T09:55:36.075462Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:36.075462Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mulberry: Empowering MLLM with o1-like Reasoning and Reflection via Collective Monte Carlo Tree Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Haocheng Feng, Huanjin Yao, Jiaxing Huang, Jingyi Zhang, Li Shen, Shunyu Liu, Wenhao Wu, Yibo Wang, Yingjie Wang, Yuxin Song","submitted_at":"2024-12-24T10:07:51Z","abstract_excerpt":"In this work, we aim to develop an MLLM that understands and solves questions by learning to create each intermediate step of the reasoning involved till the final answer. To this end, we propose Collective Monte Carlo Tree Search (CoMCTS), a new learning-to-reason method for MLLMs, which introduces the concept of collective learning into ``tree search'' for effective and efficient reasoning-path searching and learning. The core idea of CoMCTS is to leverage collective knowledge from multiple models to collaboratively conjecture, search and identify effective reasoning paths toward correct ans"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18319","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18319","created_at":"2026-07-05T09:55:36.075521+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18319v2","created_at":"2026-07-05T09:55:36.075521+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18319","created_at":"2026-07-05T09:55:36.075521+00:00"},{"alias_kind":"pith_short_12","alias_value":"P6YLOMPZUJQG","created_at":"2026-07-05T09:55:36.075521+00:00"},{"alias_kind":"pith_short_16","alias_value":"P6YLOMPZUJQGG436","created_at":"2026-07-05T09:55:36.075521+00:00"},{"alias_kind":"pith_short_8","alias_value":"P6YLOMPZ","created_at":"2026-07-05T09:55:36.075521+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22385","citing_title":"MetaPS: Adaptive Programmatic Strategy Selection for Market Agents","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08464","citing_title":"TVI-CoT: Text-Visual Interleaved Chain-of-Thought Reasoning for Multimodal Understanding","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08231","citing_title":"Test-Time Scaling in Multimodal Foundation Models: A Comprehensive Survey of Generation and Reasoning","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25571","citing_title":"AnE: Pushing the Reasoning Frontier of Multimodal LLMs via Anchor Evolution","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":235,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23678","citing_title":"Grounded Reinforcement Learning for Visual Reasoning","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17352","citing_title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22746","citing_title":"Mixture-of-Visual-Thoughts: Exploring Context-Adaptive Reasoning Mode Selection for General Visual Reasoning","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2510.14738","citing_title":"AutoRubric: Rubric-Based Generative Rewards for Faithful Multimodal Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14044","citing_title":"OmniDrive-R1: Reinforcement-driven Interleaved Multi-modal Chain-of-Thought for Trustworthy Vision-Language Autonomous Driving","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07536","citing_title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12937","citing_title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10615","citing_title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03318","citing_title":"EgoMind: Activating Spatial Cognition through Linguistic Reasoning in MLLMs","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03179","citing_title":"Understanding the Role of Hallucination in Reinforcement Post-Training of Multimodal Reasoning Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05366","citing_title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09614","citing_title":"Reflection Anchors for Propagation-Aware Visual Retention in Long-Chain Multimodal Reasoning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24339","citing_title":"See Further, Think Deeper: Advancing VLM's Reasoning Ability with Low-level Visual Cues and Reflection","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04064","citing_title":"Improving Medical VQA through Trajectory-Aware Process Supervision","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04500","citing_title":"Saliency-R1: Enforcing Interpretable and Faithful Vision-language Reasoning via Saliency-map Alignment Reward","ref_index":82,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2","json":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2.json","graph_json":"https://pith.science/api/pith-number/P6YLOMPZUJQGG436AM7DNFHHV2/graph.json","events_json":"https://pith.science/api/pith-number/P6YLOMPZUJQGG436AM7DNFHHV2/events.json","paper":"https://pith.science/paper/P6YLOMPZ"},"agent_actions":{"view_html":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2","download_json":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2.json","view_paper":"https://pith.science/paper/P6YLOMPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18319&json=true","fetch_graph":"https://pith.science/api/pith-number/P6YLOMPZUJQGG436AM7DNFHHV2/graph.json","fetch_events":"https://pith.science/api/pith-number/P6YLOMPZUJQGG436AM7DNFHHV2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2/action/storage_attestation","attest_author":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2/action/author_attestation","sign_citation":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2/action/citation_signature","submit_replication":"https://pith.science/pith/P6YLOMPZUJQGG436AM7DNFHHV2/action/replication_record"}},"created_at":"2026-07-05T09:55:36.075521+00:00","updated_at":"2026-07-05T09:55:36.075521+00:00"}