{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O64G67W25U77FHRXCJDPN7OORC","short_pith_number":"pith:O64G67W2","schema_version":"1.0","canonical_sha256":"77b86f7edaed3ff29e371246f6fdce8892d9d3bfc8dd2e1ea5dd990a443fe0d4","source":{"kind":"arxiv","id":"2402.04236","version":3},"attestation_state":"computed","paper":{"title":"CogCoM: A Visual Language Model with Chain-of-Manipulations Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Xu, Jie Tang, Ji Qi, Juanzi Li, Lei Hou, Ming Ding, Qingsong Lv, Weihan Wang, Wenyi Hong, Yushi Bai, Yuxiao Dong","submitted_at":"2024-02-06T18:43:48Z","abstract_excerpt":"Vision-Language Models (VLMs) have demonstrated their broad effectiveness thanks to extensive training in aligning visual instructions to responses. However, such training of conclusive alignment leads models to ignore essential visual reasoning, further resulting in failures in meticulous visual problems and unfaithful responses. Drawing inspiration from human cognition in solving visual problems (e.g., marking, zoom in), this paper introduces Chain of Manipulations, a mechanism that enables VLMs to solve problems step-by-step with evidence. After training, models can solve various visual pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04236","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-02-06T18:43:48Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d08e1332040ace6cf8421d3f3f8373daa2046733a5b424e166b12164ce30ed32","abstract_canon_sha256":"4264ecd8fbe52bb6e14caa2cfef4685aed2d786a0fda872254f52c9818546e92"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:57.493732Z","signature_b64":"Q+HKP0mK5VaIPb8/hj4YLo6L4oHycQQ7ZTAudPacVMaOm/4WmQ+ZjpSk8kiUmV5WRO9HEHx9zdICV5Cse562CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77b86f7edaed3ff29e371246f6fdce8892d9d3bfc8dd2e1ea5dd990a443fe0d4","last_reissued_at":"2026-07-05T10:21:57.493181Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:57.493181Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CogCoM: A Visual Language Model with Chain-of-Manipulations Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Xu, Jie Tang, Ji Qi, Juanzi Li, Lei Hou, Ming Ding, Qingsong Lv, Weihan Wang, Wenyi Hong, Yushi Bai, Yuxiao Dong","submitted_at":"2024-02-06T18:43:48Z","abstract_excerpt":"Vision-Language Models (VLMs) have demonstrated their broad effectiveness thanks to extensive training in aligning visual instructions to responses. However, such training of conclusive alignment leads models to ignore essential visual reasoning, further resulting in failures in meticulous visual problems and unfaithful responses. Drawing inspiration from human cognition in solving visual problems (e.g., marking, zoom in), this paper introduces Chain of Manipulations, a mechanism that enables VLMs to solve problems step-by-step with evidence. After training, models can solve various visual pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04236","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04236/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04236","created_at":"2026-07-05T10:21:57.493243+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04236v3","created_at":"2026-07-05T10:21:57.493243+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04236","created_at":"2026-07-05T10:21:57.493243+00:00"},{"alias_kind":"pith_short_12","alias_value":"O64G67W25U77","created_at":"2026-07-05T10:21:57.493243+00:00"},{"alias_kind":"pith_short_16","alias_value":"O64G67W25U77FHRX","created_at":"2026-07-05T10:21:57.493243+00:00"},{"alias_kind":"pith_short_8","alias_value":"O64G67W2","created_at":"2026-07-05T10:21:57.493243+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.15932","citing_title":"Beyond NL2Code: A Structured Survey of Multimodal Code Intelligence","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19342","citing_title":"Semantic-Enriched Latent Visual Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20912","citing_title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20912","citing_title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18641","citing_title":"Leveraging Latent Visual Reasoning in Silence","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19342","citing_title":"Semantic-Enriched Latent Visual Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11991","citing_title":"VGR: Visual Grounded Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20473","citing_title":"Video-ToC: Video Tree-of-Cue Reasoning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21079","citing_title":"Foveated Reasoning: Stateful, Action-based Visual Focusing for Vision-Language Models","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC","json":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC.json","graph_json":"https://pith.science/api/pith-number/O64G67W25U77FHRXCJDPN7OORC/graph.json","events_json":"https://pith.science/api/pith-number/O64G67W25U77FHRXCJDPN7OORC/events.json","paper":"https://pith.science/paper/O64G67W2"},"agent_actions":{"view_html":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC","download_json":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC.json","view_paper":"https://pith.science/paper/O64G67W2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04236&json=true","fetch_graph":"https://pith.science/api/pith-number/O64G67W25U77FHRXCJDPN7OORC/graph.json","fetch_events":"https://pith.science/api/pith-number/O64G67W25U77FHRXCJDPN7OORC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC/action/storage_attestation","attest_author":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC/action/author_attestation","sign_citation":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC/action/citation_signature","submit_replication":"https://pith.science/pith/O64G67W25U77FHRXCJDPN7OORC/action/replication_record"}},"created_at":"2026-07-05T10:21:57.493243+00:00","updated_at":"2026-07-05T10:21:57.493243+00:00"}