{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QAVTUYJHOTN3UHO2ZUNUJHLHE4","short_pith_number":"pith:QAVTUYJH","schema_version":"1.0","canonical_sha256":"802b3a612774dbba1ddacd1b449d67273e102d8d2bbea5371981135c98954fcc","source":{"kind":"arxiv","id":"2502.17422","version":1},"attestation_state":"computed","paper":{"title":"MLLMs Know Where to Look: Training-free Perception of Small Visual Details with Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Filip Ilievski, Jiarui Zhang, Mahyar Khayatkhoei, Prateek Chhikara","submitted_at":"2025-02-24T18:54:40Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have experienced rapid progress in visual recognition tasks in recent years. Given their potential integration into many critical applications, it is important to understand the limitations of their visual perception. In this work, we study whether MLLMs can perceive small visual details as effectively as large ones when answering questions about images. We observe that their performance is very sensitive to the size of the visual subject of the question, and further show that this effect is in fact causal by conducting an intervention study. Next, we s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17422","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-24T18:54:40Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"c7f2e11ec1418dc2b60141487ac9d66cb2564224f9b6d59b1ce61df6b7678465","abstract_canon_sha256":"fc01952cc7ee65baedb00f91cce804971110efff3f11b9130713f963915fd76f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:09.956643Z","signature_b64":"8ugKNxHi/N8UQhkQjRGIF82HBrqFv2HKawVxpx4/3V77127hCaP1WIZnynltCsopkKU6s7ery2CntbdC1Tq6BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"802b3a612774dbba1ddacd1b449d67273e102d8d2bbea5371981135c98954fcc","last_reissued_at":"2026-07-05T10:19:09.956148Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:09.956148Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLLMs Know Where to Look: Training-free Perception of Small Visual Details with Multimodal LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Filip Ilievski, Jiarui Zhang, Mahyar Khayatkhoei, Prateek Chhikara","submitted_at":"2025-02-24T18:54:40Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have experienced rapid progress in visual recognition tasks in recent years. Given their potential integration into many critical applications, it is important to understand the limitations of their visual perception. In this work, we study whether MLLMs can perceive small visual details as effectively as large ones when answering questions about images. We observe that their performance is very sensitive to the size of the visual subject of the question, and further show that this effect is in fact causal by conducting an intervention study. Next, we s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17422","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17422/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17422","created_at":"2026-07-05T10:19:09.956211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17422v1","created_at":"2026-07-05T10:19:09.956211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17422","created_at":"2026-07-05T10:19:09.956211+00:00"},{"alias_kind":"pith_short_12","alias_value":"QAVTUYJHOTN3","created_at":"2026-07-05T10:19:09.956211+00:00"},{"alias_kind":"pith_short_16","alias_value":"QAVTUYJHOTN3UHO2","created_at":"2026-07-05T10:19:09.956211+00:00"},{"alias_kind":"pith_short_8","alias_value":"QAVTUYJH","created_at":"2026-07-05T10:19:09.956211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05716","citing_title":"Scene Graph Thinking: Reinforcing Structured Visual Reasoning for Multimodal Large Language Models","ref_index":24,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25319","citing_title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21968","citing_title":"Look Before You Zoom: Adaptive Routing for the Resolution-Context Trade-off in Visual RAG","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18710","citing_title":"Image Prompt Reconstruction Attacks on Distributed MLLM Inference Frameworks","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08035","citing_title":"DyCo-RL: Dynamic Cross-Modal Coordination for Visual Reasoning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07861","citing_title":"The Last Visible Pixel: Probing Fine-Scale Perception in Vision-Language Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07647","citing_title":"Steer Where It Matters: Token-Level Visual-Sensitivity Steering for LVLMs Hallucination Mitigation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03376","citing_title":"P$^2$-DPO: Grounding Hallucination in Perceptual Processing via Calibration Direct Preference Optimization","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29805","citing_title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29805","citing_title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28741","citing_title":"Self-Prophetic Decoding to Unlock Visual Search in LVLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23655","citing_title":"CVSearch: Empowering Multimodal LLMs with Cognitive Visual Search for High-Resolution Image Perception","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21954","citing_title":"MLLMs Know When Before Speaking: Revealing and Recovering Temporal Grounding via Attention Cues","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14159","citing_title":"MVI-Bench: A Comprehensive Benchmark for Evaluating Robustness to Misleading Visual Inputs in LVLMs","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14184","citing_title":"Deeper Thought, Weaker Aim: Understanding and Mitigating Perceptual Impairment during Reasoning in Multimodal Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10610","citing_title":"LaSM: Layer-wise Scaling Mechanism for Defending Pop-up Attack on GUI Agents","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13415","citing_title":"Attention Grounded Enhancement for Visual Document Retrieval","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19820","citing_title":"CropVLM: Learning to Zoom for Fine-Grained Vision-Language Perception","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19219","citing_title":"Selective LoRA for Visual Tokens and Attention Heads","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17419","citing_title":"EAGLE: Expert-Augmented Attention Guidance for Tuning-Free Industrial Anomaly Detection in Multimodal Large Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27494","citing_title":"Learning to Focus and Precise Cropping: A Reinforcement Learning Framework with Information Gaps and Grounding Loss for MLLMs","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22875","citing_title":"SketchVLM: Vision language models can annotate images to explain thoughts and guide users","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08456","citing_title":"Entropy-Gradient Grounding: Training-Free Evidence Retrieval in Vision-Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02630","citing_title":"AutoFocus: Uncertainty-Aware Active Visual Search for GUI Grounding","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4","json":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4.json","graph_json":"https://pith.science/api/pith-number/QAVTUYJHOTN3UHO2ZUNUJHLHE4/graph.json","events_json":"https://pith.science/api/pith-number/QAVTUYJHOTN3UHO2ZUNUJHLHE4/events.json","paper":"https://pith.science/paper/QAVTUYJH"},"agent_actions":{"view_html":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4","download_json":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4.json","view_paper":"https://pith.science/paper/QAVTUYJH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17422&json=true","fetch_graph":"https://pith.science/api/pith-number/QAVTUYJHOTN3UHO2ZUNUJHLHE4/graph.json","fetch_events":"https://pith.science/api/pith-number/QAVTUYJHOTN3UHO2ZUNUJHLHE4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4/action/storage_attestation","attest_author":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4/action/author_attestation","sign_citation":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4/action/citation_signature","submit_replication":"https://pith.science/pith/QAVTUYJHOTN3UHO2ZUNUJHLHE4/action/replication_record"}},"created_at":"2026-07-05T10:19:09.956211+00:00","updated_at":"2026-07-05T10:19:09.956211+00:00"}