{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MEMFY45KTFSUA4CNBCZCCRU23F","short_pith_number":"pith:MEMFY45K","schema_version":"1.0","canonical_sha256":"61185c73aa996540704d08b221469ad97cf095e7a3d51690f8cefd57adc9daf8","source":{"kind":"arxiv","id":"2405.13872","version":2},"attestation_state":"computed","paper":{"title":"Image-of-Thought Prompting for Visual Reasoning Refinement in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.AI","authors_text":"Panzhong Lu, Qiji Zhou, Ruochen Zhou, Siyang Gao, Yue Zhang, Zike Hu","submitted_at":"2024-05-22T17:56:51Z","abstract_excerpt":"Recent advancements in Chain-of-Thought (CoT) and related rationale-based works have significantly improved the performance of Large Language Models (LLMs) in complex reasoning tasks. With the evolution of Multimodal Large Language Models (MLLMs), enhancing their capability to tackle complex multimodal reasoning problems is a crucial frontier. However, incorporating multimodal rationales in CoT has yet to be thoroughly investigated. We propose the Image-of-Thought (IoT) prompting method, which helps MLLMs to extract visual rationales step-by-step. Specifically, IoT prompting can automatically "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13872","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-22T17:56:51Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"e04f6f8b724cfc9dcb26364cb8f1e4f5f9aa74a25c8c504c72c7ef4bb2c896d9","abstract_canon_sha256":"b0728d79bd9efe81478609971173e96e5fd1233eb926fba3366a1d2c51832d36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:25.155133Z","signature_b64":"3WF9rqAC7QS/RABPQm/j+bLz7jcze1keUElZZo4GaPAhccEiHx6AFJuEJ5lAIEGixXJ0wplO5IHd+7GVYH0MDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61185c73aa996540704d08b221469ad97cf095e7a3d51690f8cefd57adc9daf8","last_reissued_at":"2026-07-05T08:24:25.154685Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:25.154685Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Image-of-Thought Prompting for Visual Reasoning Refinement in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.AI","authors_text":"Panzhong Lu, Qiji Zhou, Ruochen Zhou, Siyang Gao, Yue Zhang, Zike Hu","submitted_at":"2024-05-22T17:56:51Z","abstract_excerpt":"Recent advancements in Chain-of-Thought (CoT) and related rationale-based works have significantly improved the performance of Large Language Models (LLMs) in complex reasoning tasks. With the evolution of Multimodal Large Language Models (MLLMs), enhancing their capability to tackle complex multimodal reasoning problems is a crucial frontier. However, incorporating multimodal rationales in CoT has yet to be thoroughly investigated. We propose the Image-of-Thought (IoT) prompting method, which helps MLLMs to extract visual rationales step-by-step. Specifically, IoT prompting can automatically "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13872","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13872","created_at":"2026-07-05T08:24:25.154739+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13872v2","created_at":"2026-07-05T08:24:25.154739+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13872","created_at":"2026-07-05T08:24:25.154739+00:00"},{"alias_kind":"pith_short_12","alias_value":"MEMFY45KTFSU","created_at":"2026-07-05T08:24:25.154739+00:00"},{"alias_kind":"pith_short_16","alias_value":"MEMFY45KTFSUA4CN","created_at":"2026-07-05T08:24:25.154739+00:00"},{"alias_kind":"pith_short_8","alias_value":"MEMFY45K","created_at":"2026-07-05T08:24:25.154739+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08434","citing_title":"DeltaV: Thinking with Visual State Updates in Unified Large Multimodal Models","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02842","citing_title":"Spectral-Progressive Thought Flow for Lightweight Multimodal Reasoning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00384","citing_title":"VESTA: Visual Exploration with Statistical Tool Agents","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23678","citing_title":"Grounded Reinforcement Learning for Visual Reasoning","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10722","citing_title":"Bridging Brains and Machines: A Unified Frontier in Neuroscience, Artificial Intelligence, and Neuromorphic Systems","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09547","citing_title":"GoViG: Goal-Conditioned Visual Navigation Instruction Generation via Multimodal Reasoning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22746","citing_title":"Mixture-of-Visual-Thoughts: Exploring Context-Adaptive Reasoning Mode Selection for General Visual Reasoning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10941","citing_title":"Mull-Tokens: Modality-Agnostic Latent Thinking","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08156","citing_title":"LAGO: Language-Guided Adaptive Object-Region Focus for Zero-Shot Visual-Text Alignment","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02735","citing_title":"Visual Latents Know More Than They Say: Unsilencing Latent Reasoning in MLLMs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06725","citing_title":"Enhancing MLLM Spatial Understanding via Active 3D Scene Exploration for Multi-Perspective Reasoning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18562","citing_title":"AnchorSeg: Language Grounded Query Banks for Reasoning Segmentation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F","json":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F.json","graph_json":"https://pith.science/api/pith-number/MEMFY45KTFSUA4CNBCZCCRU23F/graph.json","events_json":"https://pith.science/api/pith-number/MEMFY45KTFSUA4CNBCZCCRU23F/events.json","paper":"https://pith.science/paper/MEMFY45K"},"agent_actions":{"view_html":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F","download_json":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F.json","view_paper":"https://pith.science/paper/MEMFY45K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13872&json=true","fetch_graph":"https://pith.science/api/pith-number/MEMFY45KTFSUA4CNBCZCCRU23F/graph.json","fetch_events":"https://pith.science/api/pith-number/MEMFY45KTFSUA4CNBCZCCRU23F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F/action/storage_attestation","attest_author":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F/action/author_attestation","sign_citation":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F/action/citation_signature","submit_replication":"https://pith.science/pith/MEMFY45KTFSUA4CNBCZCCRU23F/action/replication_record"}},"created_at":"2026-07-05T08:24:25.154739+00:00","updated_at":"2026-07-05T08:24:25.154739+00:00"}