{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MGYOWXK3KQGT74MWRCLQOCEWJF","short_pith_number":"pith:MGYOWXK3","schema_version":"1.0","canonical_sha256":"61b0eb5d5b540d3ff19688970708964946ad17b80baf81924c41e63ca8cfc2e1","source":{"kind":"arxiv","id":"2501.05452","version":1},"attestation_state":"computed","paper":{"title":"ReFocus: Visual Editing as a Chain of Thought for Structured Image Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Cha Zhang, Dan Roth, Dinei Florencio, Jianwei Yang, John Corring, Minqian Liu, Xingyu Fu, Yijuan Lu, Zhengyuan Yang","submitted_at":"2025-01-09T18:59:58Z","abstract_excerpt":"Structured image understanding, such as interpreting tables and charts, requires strategically refocusing across various structures and texts within an image, forming a reasoning sequence to arrive at the final answer. However, current multimodal large language models (LLMs) lack this multihop selective attention capability. In this work, we introduce ReFocus, a simple yet effective framework that equips multimodal LLMs with the ability to generate \"visual thoughts\" by performing visual editing on the input image through code, shifting and refining their visual focuses. Specifically, ReFocus e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05452","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-09T18:59:58Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"444cb463d8d990ea2be44d53e18a88d96021922cc2af1ce484500e607fc0484e","abstract_canon_sha256":"a7667624c3aad63dd8dc710b428b7bde1306bed49dd19023c08fa0a269a92414"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:10.081388Z","signature_b64":"6Q4BWO8zmEbnxowt5M9if5UxpLAE67J1iCj+NahNqUhZLbfMQFN6aTTRMpGtvhFXkmQACoo+7Ni/PB08SslcCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61b0eb5d5b540d3ff19688970708964946ad17b80baf81924c41e63ca8cfc2e1","last_reissued_at":"2026-07-05T09:59:10.080889Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:10.080889Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReFocus: Visual Editing as a Chain of Thought for Structured Image Understanding","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Cha Zhang, Dan Roth, Dinei Florencio, Jianwei Yang, John Corring, Minqian Liu, Xingyu Fu, Yijuan Lu, Zhengyuan Yang","submitted_at":"2025-01-09T18:59:58Z","abstract_excerpt":"Structured image understanding, such as interpreting tables and charts, requires strategically refocusing across various structures and texts within an image, forming a reasoning sequence to arrive at the final answer. However, current multimodal large language models (LLMs) lack this multihop selective attention capability. In this work, we introduce ReFocus, a simple yet effective framework that equips multimodal LLMs with the ability to generate \"visual thoughts\" by performing visual editing on the input image through code, shifting and refining their visual focuses. Specifically, ReFocus e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05452","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05452","created_at":"2026-07-05T09:59:10.080951+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05452v1","created_at":"2026-07-05T09:59:10.080951+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05452","created_at":"2026-07-05T09:59:10.080951+00:00"},{"alias_kind":"pith_short_12","alias_value":"MGYOWXK3KQGT","created_at":"2026-07-05T09:59:10.080951+00:00"},{"alias_kind":"pith_short_16","alias_value":"MGYOWXK3KQGT74MW","created_at":"2026-07-05T09:59:10.080951+00:00"},{"alias_kind":"pith_short_8","alias_value":"MGYOWXK3","created_at":"2026-07-05T09:59:10.080951+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24233","citing_title":"Latent Visual States for Efficient Multimodal Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04046","citing_title":"Dive into the Scene: Breaking the Perceptual Bottleneck in Vision-Language Decision Making via Focus Plan Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26014","citing_title":"STORM: Internalized Modeling for Spatial-Temporal Reasoning in Video-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00562","citing_title":"DeepLatent: Think with Images via Parallel Latent Visual Reasoning","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20912","citing_title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23678","citing_title":"Grounded Reinforcement Learning for Visual Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20912","citing_title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18740","citing_title":"Vision-OPD: Learning to See Fine Details for Multimodal LLMs via On-Policy Self-Distillation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07966","citing_title":"Visual-TableQA: Open-Domain Benchmark for Reasoning over Table Images","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08980","citing_title":"Training Multi-Image Vision Agents via End2End Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12623","citing_title":"Reasoning Within the Mind: Dynamic Multimodal Interleaving in Latent Space","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24251","citing_title":"Latent Visual Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27494","citing_title":"Learning to Focus and Precise Cropping: A Reinforcement Learning Framework with Information Gaps and Grounding Loss for MLLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03660","citing_title":"TableVision: A Large-Scale Benchmark for Spatially Grounded Reasoning over Complex Hierarchical Tables","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24625","citing_title":"Meta-CoT: Enhancing Granularity and Generalization in Image Editing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07518","citing_title":"Decompose, Look, and Reason: Reinforced Latent Reasoning for VLMs","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09757","citing_title":"MedLVR: Latent Visual Reasoning for Reliable Medical Visual Question Answering","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF","json":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF.json","graph_json":"https://pith.science/api/pith-number/MGYOWXK3KQGT74MWRCLQOCEWJF/graph.json","events_json":"https://pith.science/api/pith-number/MGYOWXK3KQGT74MWRCLQOCEWJF/events.json","paper":"https://pith.science/paper/MGYOWXK3"},"agent_actions":{"view_html":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF","download_json":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF.json","view_paper":"https://pith.science/paper/MGYOWXK3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05452&json=true","fetch_graph":"https://pith.science/api/pith-number/MGYOWXK3KQGT74MWRCLQOCEWJF/graph.json","fetch_events":"https://pith.science/api/pith-number/MGYOWXK3KQGT74MWRCLQOCEWJF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF/action/storage_attestation","attest_author":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF/action/author_attestation","sign_citation":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF/action/citation_signature","submit_replication":"https://pith.science/pith/MGYOWXK3KQGT74MWRCLQOCEWJF/action/replication_record"}},"created_at":"2026-07-05T09:59:10.080951+00:00","updated_at":"2026-07-05T09:59:10.080951+00:00"}