{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:56NVOMKSGS4G5ME526MUDF66GW","short_pith_number":"pith:56NVOMKS","schema_version":"1.0","canonical_sha256":"ef9b57315234b86eb09dd7994197de35baadc5b731e4d71ca837e06330a0b303","source":{"kind":"arxiv","id":"2402.14899","version":3},"attestation_state":"computed","paper":{"title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fan Xue, Jindong Gu, Philip Torr, Shuo Chen, Volker Tresp, Xun Xiao, Zefeng Wang, Zhen Han, Zifeng Ding","submitted_at":"2024-02-22T17:36:34Z","abstract_excerpt":"Multimodal LLMs (MLLMs) with a great ability of text and image understanding have received great attention. To achieve better reasoning with MLLMs, Chain-of-Thought (CoT) reasoning has been widely explored, which further promotes MLLMs' explainability by giving intermediate reasoning steps. Despite the strong power demonstrated by MLLMs in multimodal reasoning, recent studies show that MLLMs still suffer from adversarial images. This raises the following open questions: Does CoT also enhance the adversarial robustness of MLLMs? What do the intermediate reasoning steps of CoT entail under adver"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14899","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-22T17:36:34Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"3bc570180d6e4e0766127de3ce8bf9ec0c1ed34e0eb8ec02b867da245c0ced2f","abstract_canon_sha256":"5ddd7cbe738a99830fea7eeebd7c8ea0595cd7cbd7221490d76a70d371080d6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:09:56.450013Z","signature_b64":"3DWjt/0lSdH+nxQYDPThLYKhxv3xb+prI7q66jOXSYBtXA31ngFmp/95DUiIQvDOu/Ed+16p9nbBjYAO4PSWBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef9b57315234b86eb09dd7994197de35baadc5b731e4d71ca837e06330a0b303","last_reissued_at":"2026-07-05T09:09:56.449532Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:09:56.449532Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Fan Xue, Jindong Gu, Philip Torr, Shuo Chen, Volker Tresp, Xun Xiao, Zefeng Wang, Zhen Han, Zifeng Ding","submitted_at":"2024-02-22T17:36:34Z","abstract_excerpt":"Multimodal LLMs (MLLMs) with a great ability of text and image understanding have received great attention. To achieve better reasoning with MLLMs, Chain-of-Thought (CoT) reasoning has been widely explored, which further promotes MLLMs' explainability by giving intermediate reasoning steps. Despite the strong power demonstrated by MLLMs in multimodal reasoning, recent studies show that MLLMs still suffer from adversarial images. This raises the following open questions: Does CoT also enhance the adversarial robustness of MLLMs? What do the intermediate reasoning steps of CoT entail under adver"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14899","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14899/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14899","created_at":"2026-07-05T09:09:56.449584+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14899v3","created_at":"2026-07-05T09:09:56.449584+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14899","created_at":"2026-07-05T09:09:56.449584+00:00"},{"alias_kind":"pith_short_12","alias_value":"56NVOMKSGS4G","created_at":"2026-07-05T09:09:56.449584+00:00"},{"alias_kind":"pith_short_16","alias_value":"56NVOMKSGS4G5ME5","created_at":"2026-07-05T09:09:56.449584+00:00"},{"alias_kind":"pith_short_8","alias_value":"56NVOMKS","created_at":"2026-07-05T09:09:56.449584+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07375","citing_title":"On Adversarial Vulnerability of Vision-Language Models through the Lens of Intermediate Spectral Subspaces","ref_index":60,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11218","citing_title":"Don't Look at the Numbers: Visual Anchoring Bias and Layer-wise Representation in VLMs","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW","json":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW.json","graph_json":"https://pith.science/api/pith-number/56NVOMKSGS4G5ME526MUDF66GW/graph.json","events_json":"https://pith.science/api/pith-number/56NVOMKSGS4G5ME526MUDF66GW/events.json","paper":"https://pith.science/paper/56NVOMKS"},"agent_actions":{"view_html":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW","download_json":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW.json","view_paper":"https://pith.science/paper/56NVOMKS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14899&json=true","fetch_graph":"https://pith.science/api/pith-number/56NVOMKSGS4G5ME526MUDF66GW/graph.json","fetch_events":"https://pith.science/api/pith-number/56NVOMKSGS4G5ME526MUDF66GW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW/action/storage_attestation","attest_author":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW/action/author_attestation","sign_citation":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW/action/citation_signature","submit_replication":"https://pith.science/pith/56NVOMKSGS4G5ME526MUDF66GW/action/replication_record"}},"created_at":"2026-07-05T09:09:56.449584+00:00","updated_at":"2026-07-05T09:09:56.449584+00:00"}