{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:55M5ZALI7JNFOTTOHGRCX3EBP5","short_pith_number":"pith:55M5ZALI","schema_version":"1.0","canonical_sha256":"ef59dc8168fa5a574e6e39a22bec817f42c7e01a0f1f1b3fcf3c26c14e9ccc17","source":{"kind":"arxiv","id":"2311.06607","version":4},"attestation_state":"computed","paper":{"title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Biao Yang, Jingxu Yang, Qiang Liu, Shuo Zhang, Xiang Bai, Yabo Sun, Yuliang Liu, Zhang Li, Zhiyin Ma","submitted_at":"2023-11-11T16:37:41Z","abstract_excerpt":"Large Multimodal Models (LMMs) have shown promise in vision-language tasks but struggle with high-resolution input and detailed scene understanding. Addressing these challenges, we introduce Monkey to enhance LMM capabilities. Firstly, Monkey processes input images by dividing them into uniform patches, each matching the size (e.g., 448x448) used in the original training of the well-trained vision encoder. Equipped with individual adapter for each patch, Monkey can handle higher resolutions up to 1344x896 pixels, enabling the detailed capture of complex visual information. Secondly, it employs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.06607","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-11T16:37:41Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ce407ffad4c80ac37ce4f08bb4b34c0da30b6f273dffccdf1c301b6b14583845","abstract_canon_sha256":"a84783693a2713aef0485e2efa205e080f195b1d54e8c71818062156a2119b9d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:04.207225Z","signature_b64":"wMyPNyN+qInemHFJmQyukIxRknkjXNK+YsvBgeyBHB6cjnfN0jqqnMp8A4bx28STCgQKnv0gG0EaeDS8vlwTCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef59dc8168fa5a574e6e39a22bec817f42c7e01a0f1f1b3fcf3c26c14e9ccc17","last_reissued_at":"2026-07-05T08:59:04.206801Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:04.206801Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Biao Yang, Jingxu Yang, Qiang Liu, Shuo Zhang, Xiang Bai, Yabo Sun, Yuliang Liu, Zhang Li, Zhiyin Ma","submitted_at":"2023-11-11T16:37:41Z","abstract_excerpt":"Large Multimodal Models (LMMs) have shown promise in vision-language tasks but struggle with high-resolution input and detailed scene understanding. Addressing these challenges, we introduce Monkey to enhance LMM capabilities. Firstly, Monkey processes input images by dividing them into uniform patches, each matching the size (e.g., 448x448) used in the original training of the well-trained vision encoder. Equipped with individual adapter for each patch, Monkey can handle higher resolutions up to 1344x896 pixels, enabling the detailed capture of complex visual information. Secondly, it employs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.06607","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.06607/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.06607","created_at":"2026-07-05T08:59:04.206858+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.06607v4","created_at":"2026-07-05T08:59:04.206858+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.06607","created_at":"2026-07-05T08:59:04.206858+00:00"},{"alias_kind":"pith_short_12","alias_value":"55M5ZALI7JNF","created_at":"2026-07-05T08:59:04.206858+00:00"},{"alias_kind":"pith_short_16","alias_value":"55M5ZALI7JNFOTTO","created_at":"2026-07-05T08:59:04.206858+00:00"},{"alias_kind":"pith_short_8","alias_value":"55M5ZALI","created_at":"2026-07-05T08:59:04.206858+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21734","citing_title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","ref_index":279,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02089","citing_title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05970","citing_title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2401.16420","citing_title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10362","citing_title":"Visual Funnel: Resolving Contextual Blindness in Multimodal Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2402.00253","citing_title":"A Survey on Hallucination in Large Vision-Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2403.20330","citing_title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06281","citing_title":"MMBench: Is Your Multi-modal Model an All-around Player?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":140,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5","json":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5.json","graph_json":"https://pith.science/api/pith-number/55M5ZALI7JNFOTTOHGRCX3EBP5/graph.json","events_json":"https://pith.science/api/pith-number/55M5ZALI7JNFOTTOHGRCX3EBP5/events.json","paper":"https://pith.science/paper/55M5ZALI"},"agent_actions":{"view_html":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5","download_json":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5.json","view_paper":"https://pith.science/paper/55M5ZALI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.06607&json=true","fetch_graph":"https://pith.science/api/pith-number/55M5ZALI7JNFOTTOHGRCX3EBP5/graph.json","fetch_events":"https://pith.science/api/pith-number/55M5ZALI7JNFOTTOHGRCX3EBP5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5/action/storage_attestation","attest_author":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5/action/author_attestation","sign_citation":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5/action/citation_signature","submit_replication":"https://pith.science/pith/55M5ZALI7JNFOTTOHGRCX3EBP5/action/replication_record"}},"created_at":"2026-07-05T08:59:04.206858+00:00","updated_at":"2026-07-05T08:59:04.206858+00:00"}