{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QJH5QMMT3ZE5J2FIY6GDJEIVS7","short_pith_number":"pith:QJH5QMMT","schema_version":"1.0","canonical_sha256":"824fd83193de49d4e8a8c78c34911597f68d47c83053fb011664dc1c6820ef5d","source":{"kind":"arxiv","id":"2403.03003","version":1},"attestation_state":"computed","paper":{"title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gen Luo, Rongrong Ji, Xiaoshuai Sun, Xiawu Zheng, Yiyi Zhou, Yuxin Zhang","submitted_at":"2024-03-05T14:31:24Z","abstract_excerpt":"Despite remarkable progress, existing multimodal large language models (MLLMs) are still inferior in granular visual recognition. Contrary to previous works, we study this problem from the perspective of image resolution, and reveal that a combination of low- and high-resolution visual features can effectively mitigate this shortcoming. Based on this observation, we propose a novel and efficient method for MLLMs, termed Mixture-of-Resolution Adaptation (MRA). In particular, MRA adopts two visual pathways for images with different resolutions, where high-resolution visual information is embedde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.03003","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-05T14:31:24Z","cross_cats_sorted":[],"title_canon_sha256":"9be28452b77c27ab764e658200e02f472e9fddae093cb42f59d5d8d120b761f5","abstract_canon_sha256":"1105339342a8a96ef58ff7cec8f3f8aa9083f6a0194fd3a561b81aba53a512a6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:52:25.401080Z","signature_b64":"C6bc8MVXPM0fzUYCHbOroKVeJXnhiDwVomD2+lX8YQh15STOwcXQlUmEghgOjtZmKORjqIcr3sDanEqTSUHoCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"824fd83193de49d4e8a8c78c34911597f68d47c83053fb011664dc1c6820ef5d","last_reissued_at":"2026-07-05T07:52:25.400727Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:52:25.400727Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Gen Luo, Rongrong Ji, Xiaoshuai Sun, Xiawu Zheng, Yiyi Zhou, Yuxin Zhang","submitted_at":"2024-03-05T14:31:24Z","abstract_excerpt":"Despite remarkable progress, existing multimodal large language models (MLLMs) are still inferior in granular visual recognition. Contrary to previous works, we study this problem from the perspective of image resolution, and reveal that a combination of low- and high-resolution visual features can effectively mitigate this shortcoming. Based on this observation, we propose a novel and efficient method for MLLMs, termed Mixture-of-Resolution Adaptation (MRA). In particular, MRA adopts two visual pathways for images with different resolutions, where high-resolution visual information is embedde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.03003","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.03003/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.03003","created_at":"2026-07-05T07:52:25.400784+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.03003v1","created_at":"2026-07-05T07:52:25.400784+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.03003","created_at":"2026-07-05T07:52:25.400784+00:00"},{"alias_kind":"pith_short_12","alias_value":"QJH5QMMT3ZE5","created_at":"2026-07-05T07:52:25.400784+00:00"},{"alias_kind":"pith_short_16","alias_value":"QJH5QMMT3ZE5J2FI","created_at":"2026-07-05T07:52:25.400784+00:00"},{"alias_kind":"pith_short_8","alias_value":"QJH5QMMT","created_at":"2026-07-05T07:52:25.400784+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10651","citing_title":"Kwai Keye-VL-2.0 Technical Report","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03879","citing_title":"Beyond Encoder Accumulation: Measuring Encoder Roles in Multi-Encoder VLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24675","citing_title":"VaaWIT: Visual-Aware Adaptation of Large Language Models for Multilingual Web Image Translation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22911","citing_title":"ForestPrune: High-ratio Visual Token Compression for Video Multimodal Large Language Models via Spatial-Temporal Forest Modeling","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12508","citing_title":"From Attenuation to Attention: Variational Information Flow Manipulation for Fine-Grained Visual Perception","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06912","citing_title":"Q-Zoom: Query-Aware Adaptive Perception for Efficient Multimodal Large Language Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13565","citing_title":"UHR-BAT: Budget-Aware Token Compression Vision-Language model for Ultra-High-Resolution Remote Sensing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18265","citing_title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15670","citing_title":"PixDLM: A Dual-Path Multimodal Language Model for UAV Reasoning Segmentation","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7","json":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7.json","graph_json":"https://pith.science/api/pith-number/QJH5QMMT3ZE5J2FIY6GDJEIVS7/graph.json","events_json":"https://pith.science/api/pith-number/QJH5QMMT3ZE5J2FIY6GDJEIVS7/events.json","paper":"https://pith.science/paper/QJH5QMMT"},"agent_actions":{"view_html":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7","download_json":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7.json","view_paper":"https://pith.science/paper/QJH5QMMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.03003&json=true","fetch_graph":"https://pith.science/api/pith-number/QJH5QMMT3ZE5J2FIY6GDJEIVS7/graph.json","fetch_events":"https://pith.science/api/pith-number/QJH5QMMT3ZE5J2FIY6GDJEIVS7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7/action/storage_attestation","attest_author":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7/action/author_attestation","sign_citation":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7/action/citation_signature","submit_replication":"https://pith.science/pith/QJH5QMMT3ZE5J2FIY6GDJEIVS7/action/replication_record"}},"created_at":"2026-07-05T07:52:25.400784+00:00","updated_at":"2026-07-05T07:52:25.400784+00:00"}