{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QX5VEDMDUQZ6HS3EMYG7GIY27A","short_pith_number":"pith:QX5VEDMD","schema_version":"1.0","canonical_sha256":"85fb520d83a433e3cb64660df3231af81557db2de3e0920dc03b874a3ee17444","source":{"kind":"arxiv","id":"2403.16999","version":3},"attestation_state":"computed","paper":{"title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guanglu Song, Han Xiao, Hao Shao, Hongsheng Li, Letian Wang, Shengju Qian, Yu Liu, Zhuofan Zong","submitted_at":"2024-03-25T17:59:23Z","abstract_excerpt":"Multi-Modal Large Language Models (MLLMs) have demonstrated impressive performance in various VQA tasks. However, they often lack interpretability and struggle with complex visual inputs, especially when the resolution of the input image is high or when the interested region that could provide key information for answering the question is small. To address these challenges, we collect and introduce the large-scale Visual CoT dataset comprising 438k question-answer pairs, annotated with intermediate bounding boxes highlighting key regions essential for answering the questions. Additionally, abo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.16999","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-25T17:59:23Z","cross_cats_sorted":[],"title_canon_sha256":"0cd95bea3494f033bd819b73ff15c74b45817e14ebdd65aab7bef8f59fb810d9","abstract_canon_sha256":"dde7084216d0c19acf3c7253c64ca0ef97b6bf16211ee578d8ee8fc8c1eb7acc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:11.613018Z","signature_b64":"GNe/LoUq8dVRXY/kvU38a3c3Y4ZWZCr/EdY6uXw2iOsELQTG4l7eJLgZz6jlTlOO8IE8e+NYvvZ8saGFrBJHAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85fb520d83a433e3cb64660df3231af81557db2de3e0920dc03b874a3ee17444","last_reissued_at":"2026-07-05T09:30:11.612589Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:11.612589Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guanglu Song, Han Xiao, Hao Shao, Hongsheng Li, Letian Wang, Shengju Qian, Yu Liu, Zhuofan Zong","submitted_at":"2024-03-25T17:59:23Z","abstract_excerpt":"Multi-Modal Large Language Models (MLLMs) have demonstrated impressive performance in various VQA tasks. However, they often lack interpretability and struggle with complex visual inputs, especially when the resolution of the input image is high or when the interested region that could provide key information for answering the question is small. To address these challenges, we collect and introduce the large-scale Visual CoT dataset comprising 438k question-answer pairs, annotated with intermediate bounding boxes highlighting key regions essential for answering the questions. Additionally, abo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.16999","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.16999/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.16999","created_at":"2026-07-05T09:30:11.612645+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.16999v3","created_at":"2026-07-05T09:30:11.612645+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.16999","created_at":"2026-07-05T09:30:11.612645+00:00"},{"alias_kind":"pith_short_12","alias_value":"QX5VEDMDUQZ6","created_at":"2026-07-05T09:30:11.612645+00:00"},{"alias_kind":"pith_short_16","alias_value":"QX5VEDMDUQZ6HS3E","created_at":"2026-07-05T09:30:11.612645+00:00"},{"alias_kind":"pith_short_8","alias_value":"QX5VEDMD","created_at":"2026-07-05T09:30:11.612645+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02089","citing_title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00461","citing_title":"Multimodal Continuous Reasoning via Asymmetric Mutual Variational Learning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25524","citing_title":"ProSR: Process-Shaped Spatial Reasoning for Reliable Chain-of-Thought in VLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28741","citing_title":"Self-Prophetic Decoding to Unlock Visual Search in LVLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16410","citing_title":"Test-Time Hinting for Black-Box Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.10026","citing_title":"LaV-CoT: Language-Aware Visual CoT with Multi-Aspect Reward Optimization for Real-World Multilingual VQA","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22746","citing_title":"Mixture-of-Visual-Thoughts: Exploring Context-Adaptive Reasoning Mode Selection for General Visual Reasoning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22020","citing_title":"CoT-VLA: Visual Chain-of-Thought Reasoning for Vision-Language-Action Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17800","citing_title":"ReFineVLA: Multimodal Reasoning-Aware Generalist Robotic Policies via Teacher-Guided Fine-Tuning","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A","json":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A.json","graph_json":"https://pith.science/api/pith-number/QX5VEDMDUQZ6HS3EMYG7GIY27A/graph.json","events_json":"https://pith.science/api/pith-number/QX5VEDMDUQZ6HS3EMYG7GIY27A/events.json","paper":"https://pith.science/paper/QX5VEDMD"},"agent_actions":{"view_html":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A","download_json":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A.json","view_paper":"https://pith.science/paper/QX5VEDMD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.16999&json=true","fetch_graph":"https://pith.science/api/pith-number/QX5VEDMDUQZ6HS3EMYG7GIY27A/graph.json","fetch_events":"https://pith.science/api/pith-number/QX5VEDMDUQZ6HS3EMYG7GIY27A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A/action/storage_attestation","attest_author":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A/action/author_attestation","sign_citation":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A/action/citation_signature","submit_replication":"https://pith.science/pith/QX5VEDMDUQZ6HS3EMYG7GIY27A/action/replication_record"}},"created_at":"2026-07-05T09:30:11.612645+00:00","updated_at":"2026-07-05T09:30:11.612645+00:00"}