{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ATLIRKDSORRGYYC7F4PCP63ZZS","short_pith_number":"pith:ATLIRKDS","schema_version":"1.0","canonical_sha256":"04d688a87274626c605f2f1e27fb79ccbb8edb56dacc94f79b137da59ed03e49","source":{"kind":"arxiv","id":"2311.09241","version":1},"attestation_state":"computed","paper":{"title":"Chain of Images for Intuitively Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fanxu Meng, Haotong Yang, Muhan Zhang, Yiding Wang","submitted_at":"2023-11-09T11:14:51Z","abstract_excerpt":"The human brain is naturally equipped to comprehend and interpret visual information rapidly. When confronted with complex problems or concepts, we use flowcharts, sketches, and diagrams to aid our thought process. Leveraging this inherent ability can significantly enhance logical reasoning. However, current Large Language Models (LLMs) do not utilize such visual intuition to help their thinking. Even the most advanced version language models (e.g., GPT-4V and LLaVA) merely align images into textual space, which means their reasoning processes remain purely verbal. To mitigate such limitations"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09241","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-09T11:14:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6b9598ac0a083c565fe84134cc403e60fb5d34269683e5f283bb3eebbf150c63","abstract_canon_sha256":"c7c734ab46939a3a25cda09df17ba6c094935540288f3b8cf8c65f24f8a283ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:13:11.802735Z","signature_b64":"vqRQomg7RHPWO+EQJq6YT/GHhFUp6LGHY01N+aGbSlGVQS02yWHI0DZD4dtvqlJyEvBvUkd1hOpu/9rCPrl/Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04d688a87274626c605f2f1e27fb79ccbb8edb56dacc94f79b137da59ed03e49","last_reissued_at":"2026-07-05T07:13:11.802266Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:13:11.802266Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Chain of Images for Intuitively Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fanxu Meng, Haotong Yang, Muhan Zhang, Yiding Wang","submitted_at":"2023-11-09T11:14:51Z","abstract_excerpt":"The human brain is naturally equipped to comprehend and interpret visual information rapidly. When confronted with complex problems or concepts, we use flowcharts, sketches, and diagrams to aid our thought process. Leveraging this inherent ability can significantly enhance logical reasoning. However, current Large Language Models (LLMs) do not utilize such visual intuition to help their thinking. Even the most advanced version language models (e.g., GPT-4V and LLaVA) merely align images into textual space, which means their reasoning processes remain purely verbal. To mitigate such limitations"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09241","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09241/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09241","created_at":"2026-07-05T07:13:11.802323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09241v1","created_at":"2026-07-05T07:13:11.802323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09241","created_at":"2026-07-05T07:13:11.802323+00:00"},{"alias_kind":"pith_short_12","alias_value":"ATLIRKDSORRG","created_at":"2026-07-05T07:13:11.802323+00:00"},{"alias_kind":"pith_short_16","alias_value":"ATLIRKDSORRGYYC7","created_at":"2026-07-05T07:13:11.802323+00:00"},{"alias_kind":"pith_short_8","alias_value":"ATLIRKDS","created_at":"2026-07-05T07:13:11.802323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS","json":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS.json","graph_json":"https://pith.science/api/pith-number/ATLIRKDSORRGYYC7F4PCP63ZZS/graph.json","events_json":"https://pith.science/api/pith-number/ATLIRKDSORRGYYC7F4PCP63ZZS/events.json","paper":"https://pith.science/paper/ATLIRKDS"},"agent_actions":{"view_html":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS","download_json":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS.json","view_paper":"https://pith.science/paper/ATLIRKDS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09241&json=true","fetch_graph":"https://pith.science/api/pith-number/ATLIRKDSORRGYYC7F4PCP63ZZS/graph.json","fetch_events":"https://pith.science/api/pith-number/ATLIRKDSORRGYYC7F4PCP63ZZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS/action/storage_attestation","attest_author":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS/action/author_attestation","sign_citation":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS/action/citation_signature","submit_replication":"https://pith.science/pith/ATLIRKDSORRGYYC7F4PCP63ZZS/action/replication_record"}},"created_at":"2026-07-05T07:13:11.802323+00:00","updated_at":"2026-07-05T07:13:11.802323+00:00"}