{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VUWH5DO632VWXI7XELQVCT6XQK","short_pith_number":"pith:VUWH5DO6","schema_version":"1.0","canonical_sha256":"ad2c7e8ddedeab6ba3f722e1514fd782af1e4c59c506d0aec20fbc93d97bdbc8","source":{"kind":"arxiv","id":"2501.01904","version":2},"attestation_state":"computed","paper":{"title":"Virgo: A Preliminary Exploration on Reproducing o1-like MLLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingning Wang, Ji-Rong Wen, Wayne Xin Zhao, Weipeng Chen, Yifan Du, Yifan Li, Yuqi Huo, Zheng Liu, Zhongyuan Wang, Zikang Liu","submitted_at":"2025-01-03T17:14:16Z","abstract_excerpt":"Recently, slow-thinking reasoning systems, built upon large language models (LLMs), have garnered widespread attention by scaling the thinking time during inference. There is also growing interest in adapting this capability to multimodal large language models (MLLMs). Given that MLLMs handle more complex data semantics across different modalities, it is intuitively more challenging to implement multimodal slow-thinking systems.\n  To address this issue, in this paper, we explore a straightforward approach by fine-tuning a capable MLLM with a small amount of textual long-form thought data, resu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.01904","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-03T17:14:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"84ca0ef266a641e88842bd70425050200b3c534bf7ea27e5494e8a289fc720f2","abstract_canon_sha256":"c7a48a45d52693c0fd427f439b80cf8192218576ee4676d5b514a2190280bd17"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:51.044408Z","signature_b64":"8RDtXkaEnBybbHYnlIQp6EHa8gDlsteG+mPgw9n4BP/8jVR5KBxGVpKMyqeGykjOfqnLNqyOZyTtwiTowsX7Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad2c7e8ddedeab6ba3f722e1514fd782af1e4c59c506d0aec20fbc93d97bdbc8","last_reissued_at":"2026-07-05T10:09:51.043902Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:51.043902Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Virgo: A Preliminary Exploration on Reproducing o1-like MLLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bingning Wang, Ji-Rong Wen, Wayne Xin Zhao, Weipeng Chen, Yifan Du, Yifan Li, Yuqi Huo, Zheng Liu, Zhongyuan Wang, Zikang Liu","submitted_at":"2025-01-03T17:14:16Z","abstract_excerpt":"Recently, slow-thinking reasoning systems, built upon large language models (LLMs), have garnered widespread attention by scaling the thinking time during inference. There is also growing interest in adapting this capability to multimodal large language models (MLLMs). Given that MLLMs handle more complex data semantics across different modalities, it is intuitively more challenging to implement multimodal slow-thinking systems.\n  To address this issue, in this paper, we explore a straightforward approach by fine-tuning a capable MLLM with a small amount of textual long-form thought data, resu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.01904","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.01904/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.01904","created_at":"2026-07-05T10:09:51.043964+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.01904v2","created_at":"2026-07-05T10:09:51.043964+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.01904","created_at":"2026-07-05T10:09:51.043964+00:00"},{"alias_kind":"pith_short_12","alias_value":"VUWH5DO632VW","created_at":"2026-07-05T10:09:51.043964+00:00"},{"alias_kind":"pith_short_16","alias_value":"VUWH5DO632VWXI7X","created_at":"2026-07-05T10:09:51.043964+00:00"},{"alias_kind":"pith_short_8","alias_value":"VUWH5DO6","created_at":"2026-07-05T10:09:51.043964+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01707","citing_title":"LASER: A Corrective Lens for LVLMs via Visual Attention Preservation and Sink Suppression","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17352","citing_title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22746","citing_title":"Mixture-of-Visual-Thoughts: Exploring Context-Adaptive Reasoning Mode Selection for General Visual Reasoning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23322","citing_title":"Mitigating Visual Context Degradation in Large Multimodal Models: A Training-Free Decoupled Agentic Framework","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14044","citing_title":"OmniDrive-R1: Reinforcement-driven Interleaved Multi-modal Chain-of-Thought for Trustworthy Vision-Language Autonomous Driving","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04442","citing_title":"Addressing Overthinking in Large Vision-Language Models via Gated Perception-Reasoning Optimization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2504.01805","citing_title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":169,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24339","citing_title":"See Further, Think Deeper: Advancing VLM's Reasoning Ability with Low-level Visual Cues and Reflection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12890","citing_title":"Towards Long-horizon Agentic Multimodal Search","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK","json":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK.json","graph_json":"https://pith.science/api/pith-number/VUWH5DO632VWXI7XELQVCT6XQK/graph.json","events_json":"https://pith.science/api/pith-number/VUWH5DO632VWXI7XELQVCT6XQK/events.json","paper":"https://pith.science/paper/VUWH5DO6"},"agent_actions":{"view_html":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK","download_json":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK.json","view_paper":"https://pith.science/paper/VUWH5DO6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.01904&json=true","fetch_graph":"https://pith.science/api/pith-number/VUWH5DO632VWXI7XELQVCT6XQK/graph.json","fetch_events":"https://pith.science/api/pith-number/VUWH5DO632VWXI7XELQVCT6XQK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK/action/storage_attestation","attest_author":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK/action/author_attestation","sign_citation":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK/action/citation_signature","submit_replication":"https://pith.science/pith/VUWH5DO632VWXI7XELQVCT6XQK/action/replication_record"}},"created_at":"2026-07-05T10:09:51.043964+00:00","updated_at":"2026-07-05T10:09:51.043964+00:00"}