{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:A5JWW6FMXC54LM2AV6HM2QX2RO","short_pith_number":"pith:A5JWW6FM","schema_version":"1.0","canonical_sha256":"07536b78acb8bbc5b340af8ecd42fa8b8605ae8d7ab466166579cb34327169ce","source":{"kind":"arxiv","id":"2504.18397","version":2},"attestation_state":"computed","paper":{"title":"Unsupervised Visual Chain-of-Thought Reasoning via Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Beier Zhu, Hanwang Zhang, Kesen Zhao, Qianru Sun","submitted_at":"2025-04-25T14:48:18Z","abstract_excerpt":"Chain-of-thought (CoT) reasoning greatly improves the interpretability and problem-solving abilities of multimodal large language models (MLLMs). However, existing approaches are focused on text CoT, limiting their ability to leverage visual cues. Visual CoT remains underexplored, and the only work is based on supervised fine-tuning (SFT) that relies on extensive labeled bounding-box data and is hard to generalize to unseen cases. In this paper, we introduce Unsupervised Visual CoT (UV-CoT), a novel framework for image-level CoT reasoning via preference optimization. UV-CoT performs preference"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.18397","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-25T14:48:18Z","cross_cats_sorted":[],"title_canon_sha256":"a960bf62eeef491db75a061d5c1e7c77b1f47167d035db84307d99ea56dab309","abstract_canon_sha256":"39dc8e5531a6a398c1215ccc246f9236fe8a642e9f0f7cee39f1dcfa8945865a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:37:18.862187Z","signature_b64":"m+wLMAMcpycx9412Z9CWb/C3+p5/b3xqVTFVcieHPO4C2BqcqI+4UmqD4Ous90CI0tJMPKr5+7+KVKhY952XAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07536b78acb8bbc5b340af8ecd42fa8b8605ae8d7ab466166579cb34327169ce","last_reissued_at":"2026-07-05T11:37:18.861697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:37:18.861697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unsupervised Visual Chain-of-Thought Reasoning via Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Beier Zhu, Hanwang Zhang, Kesen Zhao, Qianru Sun","submitted_at":"2025-04-25T14:48:18Z","abstract_excerpt":"Chain-of-thought (CoT) reasoning greatly improves the interpretability and problem-solving abilities of multimodal large language models (MLLMs). However, existing approaches are focused on text CoT, limiting their ability to leverage visual cues. Visual CoT remains underexplored, and the only work is based on supervised fine-tuning (SFT) that relies on extensive labeled bounding-box data and is hard to generalize to unseen cases. In this paper, we introduce Unsupervised Visual CoT (UV-CoT), a novel framework for image-level CoT reasoning via preference optimization. UV-CoT performs preference"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.18397","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.18397/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.18397","created_at":"2026-07-05T11:37:18.861753+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.18397v2","created_at":"2026-07-05T11:37:18.861753+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.18397","created_at":"2026-07-05T11:37:18.861753+00:00"},{"alias_kind":"pith_short_12","alias_value":"A5JWW6FMXC54","created_at":"2026-07-05T11:37:18.861753+00:00"},{"alias_kind":"pith_short_16","alias_value":"A5JWW6FMXC54LM2A","created_at":"2026-07-05T11:37:18.861753+00:00"},{"alias_kind":"pith_short_8","alias_value":"A5JWW6FM","created_at":"2026-07-05T11:37:18.861753+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.15879","citing_title":"GRIT: Teaching MLLMs to Think with Images","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00054","citing_title":"HiDe: Rethinking The Zoom-IN method in High Resolution MLLMs via Hierarchical Decoupling","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.10026","citing_title":"LaV-CoT: Language-Aware Visual CoT with Multi-Aspect Reward Optimization for Real-World Multilingual VQA","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19820","citing_title":"CropVLM: Learning to Zoom for Fine-Grained Vision-Language Perception","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO","json":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO.json","graph_json":"https://pith.science/api/pith-number/A5JWW6FMXC54LM2AV6HM2QX2RO/graph.json","events_json":"https://pith.science/api/pith-number/A5JWW6FMXC54LM2AV6HM2QX2RO/events.json","paper":"https://pith.science/paper/A5JWW6FM"},"agent_actions":{"view_html":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO","download_json":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO.json","view_paper":"https://pith.science/paper/A5JWW6FM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.18397&json=true","fetch_graph":"https://pith.science/api/pith-number/A5JWW6FMXC54LM2AV6HM2QX2RO/graph.json","fetch_events":"https://pith.science/api/pith-number/A5JWW6FMXC54LM2AV6HM2QX2RO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO/action/storage_attestation","attest_author":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO/action/author_attestation","sign_citation":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO/action/citation_signature","submit_replication":"https://pith.science/pith/A5JWW6FMXC54LM2AV6HM2QX2RO/action/replication_record"}},"created_at":"2026-07-05T11:37:18.861753+00:00","updated_at":"2026-07-05T11:37:18.861753+00:00"}