{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RLJ5UB2NJWLKC2BNKUNPRY2W5F","short_pith_number":"pith:RLJ5UB2N","schema_version":"1.0","canonical_sha256":"8ad3da074d4d96a1682d551af8e356e951919c48c5fa2964789a4d71acdeec5f","source":{"kind":"arxiv","id":"2507.03019","version":1},"attestation_state":"computed","paper":{"title":"Look-Back: Implicit Visual Re-focusing in MLLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bin Lin, Li Yuan, Shuo Yang, Yang Ye, Yuwei Niu, Yuyang Liu","submitted_at":"2025-07-02T14:59:35Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved remarkable progress in multimodal reasoning. However, they often excessively rely on textual information during the later stages of inference, neglecting the crucial integration of visual input. Current methods typically address this by explicitly injecting visual information to guide the reasoning process. In this work, through an analysis of MLLM attention patterns, we made an intriguing observation: with appropriate guidance, MLLMs can spontaneously re-focus their attention on visual inputs during the later stages of reasoning, even wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.03019","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-02T14:59:35Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a4698fb6db40858d5681e6fb3dc87aaa173ceab010fe6b09ceaa69f15881d815","abstract_canon_sha256":"3b1dddd4f9e77b6bee519ada7c78066d8d111824b123020232e021b522c7f42a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:50.831987Z","signature_b64":"GwnkX3MZOKrU6mEfO/HXJGYqw5OOFJft/P8kba8xNwVQJIBCuzsKGzOjoSY+0FeM4gKyNRvMVmeFdnQq5YjUDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8ad3da074d4d96a1682d551af8e356e951919c48c5fa2964789a4d71acdeec5f","last_reissued_at":"2026-07-05T11:31:50.831479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:50.831479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Look-Back: Implicit Visual Re-focusing in MLLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bin Lin, Li Yuan, Shuo Yang, Yang Ye, Yuwei Niu, Yuyang Liu","submitted_at":"2025-07-02T14:59:35Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have achieved remarkable progress in multimodal reasoning. However, they often excessively rely on textual information during the later stages of inference, neglecting the crucial integration of visual input. Current methods typically address this by explicitly injecting visual information to guide the reasoning process. In this work, through an analysis of MLLM attention patterns, we made an intriguing observation: with appropriate guidance, MLLMs can spontaneously re-focus their attention on visual inputs during the later stages of reasoning, even wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.03019","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.03019/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.03019","created_at":"2026-07-05T11:31:50.831541+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.03019v1","created_at":"2026-07-05T11:31:50.831541+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.03019","created_at":"2026-07-05T11:31:50.831541+00:00"},{"alias_kind":"pith_short_12","alias_value":"RLJ5UB2NJWLK","created_at":"2026-07-05T11:31:50.831541+00:00"},{"alias_kind":"pith_short_16","alias_value":"RLJ5UB2NJWLKC2BN","created_at":"2026-07-05T11:31:50.831541+00:00"},{"alias_kind":"pith_short_8","alias_value":"RLJ5UB2N","created_at":"2026-07-05T11:31:50.831541+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05716","citing_title":"Scene Graph Thinking: Reinforcing Structured Visual Reasoning for Multimodal Large Language Models","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26196","citing_title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","ref_index":212,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01707","citing_title":"LASER: A Corrective Lens for LVLMs via Visual Attention Preservation and Sink Suppression","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01558","citing_title":"Attention-guided Fine-tuning of Multimodal Large Language Models Improves Chain-of-Thought Reasoning","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18359","citing_title":"RAVE: Re-Allocating Visual Attention in Large Multimodal Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31457","citing_title":"VisionPulse: Dynamic Visual Sparsity for Efficient Multimodal Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15951","citing_title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18903","citing_title":"Reasoning Portability: Guiding Continual Learning for MLLMs in the RLVR Era","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18359","citing_title":"RAVE: Re-Allocating Visual Attention in Large Multimodal Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13026","citing_title":"REVISOR: Beyond Textual Reflection, Towards Multimodal Introspective Reasoning in Long-Form Video Understanding","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19945","citing_title":"Visual Reasoning through Tool-supervised Reinforcement Learning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10500","citing_title":"Visual Enhanced Depth Scaling for Multimodal Latent Reasoning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10071","citing_title":"Spotlight and Shadow: Attention-Guided Dual-Anchor Introspective Decoding for MLLM Hallucination Mitigation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F","json":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F.json","graph_json":"https://pith.science/api/pith-number/RLJ5UB2NJWLKC2BNKUNPRY2W5F/graph.json","events_json":"https://pith.science/api/pith-number/RLJ5UB2NJWLKC2BNKUNPRY2W5F/events.json","paper":"https://pith.science/paper/RLJ5UB2N"},"agent_actions":{"view_html":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F","download_json":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F.json","view_paper":"https://pith.science/paper/RLJ5UB2N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.03019&json=true","fetch_graph":"https://pith.science/api/pith-number/RLJ5UB2NJWLKC2BNKUNPRY2W5F/graph.json","fetch_events":"https://pith.science/api/pith-number/RLJ5UB2NJWLKC2BNKUNPRY2W5F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F/action/storage_attestation","attest_author":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F/action/author_attestation","sign_citation":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F/action/citation_signature","submit_replication":"https://pith.science/pith/RLJ5UB2NJWLKC2BNKUNPRY2W5F/action/replication_record"}},"created_at":"2026-07-05T11:31:50.831541+00:00","updated_at":"2026-07-05T11:31:50.831541+00:00"}