{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HSDCS5LQGID2X2XHLJPQ7ORN6F","short_pith_number":"pith:HSDCS5LQ","schema_version":"1.0","canonical_sha256":"3c862975703207abeae75a5f0fba2df158305a5343f0937281eacd6c5d182b58","source":{"kind":"arxiv","id":"2408.15556","version":1},"attestation_state":"computed","paper":{"title":"Divide, Conquer and Combine: A Training-Free Framework for High-Resolution Image Perception in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Liang Ding, Li Shen, Minyan Zeng, Wenbin Wang, Xiabin Zhou, Yong Luo","submitted_at":"2024-08-28T06:09:02Z","abstract_excerpt":"Multimodal large language models (MLLMs) have experienced significant advancements recently, but still struggle to recognize and interpret intricate details in high-resolution (HR) images effectively. While state-of-the-art (SOTA) MLLMs claim to process images at 4K resolution, existing MLLM benchmarks only support up to 2K, leaving the capabilities of SOTA models on true HR images largely untested. Furthermore, existing methods for enhancing HR image perception in MLLMs rely on computationally expensive visual instruction tuning. To address these limitations, we introduce HR-Bench, the first "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.15556","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-28T06:09:02Z","cross_cats_sorted":[],"title_canon_sha256":"250ff6f1609f9c10e82ac6c665e805a09718f02a7424526c461fa907ec5cc6d5","abstract_canon_sha256":"5647f4eac5facdc81df69e1871b0145d54e0314e536b255c5ae375ab934302ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:13.688874Z","signature_b64":"FQmRUd62FRIrRSw7HxJ/EDaTeQdQQhzP5Yqk2PTRszd5Dj3gvxA9BapZ9243sgsD+6SyVu4v9qklSB3bYIHlAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c862975703207abeae75a5f0fba2df158305a5343f0937281eacd6c5d182b58","last_reissued_at":"2026-07-05T09:00:13.688389Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:13.688389Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Divide, Conquer and Combine: A Training-Free Framework for High-Resolution Image Perception in Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dacheng Tao, Liang Ding, Li Shen, Minyan Zeng, Wenbin Wang, Xiabin Zhou, Yong Luo","submitted_at":"2024-08-28T06:09:02Z","abstract_excerpt":"Multimodal large language models (MLLMs) have experienced significant advancements recently, but still struggle to recognize and interpret intricate details in high-resolution (HR) images effectively. While state-of-the-art (SOTA) MLLMs claim to process images at 4K resolution, existing MLLM benchmarks only support up to 2K, leaving the capabilities of SOTA models on true HR images largely untested. Furthermore, existing methods for enhancing HR image perception in MLLMs rely on computationally expensive visual instruction tuning. To address these limitations, we introduce HR-Bench, the first "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.15556","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.15556/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.15556","created_at":"2026-07-05T09:00:13.688449+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.15556v1","created_at":"2026-07-05T09:00:13.688449+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.15556","created_at":"2026-07-05T09:00:13.688449+00:00"},{"alias_kind":"pith_short_12","alias_value":"HSDCS5LQGID2","created_at":"2026-07-05T09:00:13.688449+00:00"},{"alias_kind":"pith_short_16","alias_value":"HSDCS5LQGID2X2XH","created_at":"2026-07-05T09:00:13.688449+00:00"},{"alias_kind":"pith_short_8","alias_value":"HSDCS5LQ","created_at":"2026-07-05T09:00:13.688449+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05576","citing_title":"UltraVR: A Diagnostic Ultra-Resolution Image-VQA Benchmark for Evidence-Grounded Reasoning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25863","citing_title":"MCMit: Mid-Circuit Measurement Error Mitigation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20278","citing_title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20278","citing_title":"ClaimDiff-RL: Fine-Grained Caption Reinforcement Learning through Visual Claim Comparison","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18603","citing_title":"Starve to Perceive: Taming Lazy Perception in VLMs with Constrained Visual Bandwidth","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15436","citing_title":"Adaptive Chain-of-Focus Reasoning via Dynamic Visual Search and Zooming for Efficient VLMs","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25855","citing_title":"SIEVES: Selective Prediction Generalizes through Visual Evidence Scoring","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25855","citing_title":"SIEVES: Selective Prediction Generalizes through Visual Evidence Scoring","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F","json":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F.json","graph_json":"https://pith.science/api/pith-number/HSDCS5LQGID2X2XHLJPQ7ORN6F/graph.json","events_json":"https://pith.science/api/pith-number/HSDCS5LQGID2X2XHLJPQ7ORN6F/events.json","paper":"https://pith.science/paper/HSDCS5LQ"},"agent_actions":{"view_html":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F","download_json":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F.json","view_paper":"https://pith.science/paper/HSDCS5LQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.15556&json=true","fetch_graph":"https://pith.science/api/pith-number/HSDCS5LQGID2X2XHLJPQ7ORN6F/graph.json","fetch_events":"https://pith.science/api/pith-number/HSDCS5LQGID2X2XHLJPQ7ORN6F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F/action/storage_attestation","attest_author":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F/action/author_attestation","sign_citation":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F/action/citation_signature","submit_replication":"https://pith.science/pith/HSDCS5LQGID2X2XHLJPQ7ORN6F/action/replication_record"}},"created_at":"2026-07-05T09:00:13.688449+00:00","updated_at":"2026-07-05T09:00:13.688449+00:00"}