{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EGO2DI5W6N63OAVOZBFO4F2MUU","short_pith_number":"pith:EGO2DI5W","schema_version":"1.0","canonical_sha256":"219da1a3b6f37db702aec84aee174ca50cc73fde560ddca53bfec0efd38d3ae9","source":{"kind":"arxiv","id":"2404.06512","version":1},"attestation_state":"computed","paper":{"title":"InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Wang, Conghui He, Dahua Lin, Hang Yan, Haodong Duan, Jiaqi Wang, Jifeng Dai, Jingwen Li, Kai Chen, Linke Ouyang, Pan Zhang, Songyang Zhang, Wei Li, Wenhai Wang, Wenwei Zhang, Xiaoyi Dong, Xingcheng Zhang, Xinyue Zhang, Yang Gao, Yining Li, Yuhang Cao, Yuhang Zang, Yu Qiao, Zhe Chen","submitted_at":"2024-04-09T17:59:32Z","abstract_excerpt":"The Large Vision-Language Model (LVLM) field has seen significant advancements, yet its progression has been hindered by challenges in comprehending fine-grained visual content due to limited resolution. Recent efforts have aimed to enhance the high-resolution understanding capabilities of LVLMs, yet they remain capped at approximately 1500 x 1500 pixels and constrained to a relatively narrow resolution range. This paper represents InternLM-XComposer2-4KHD, a groundbreaking exploration into elevating LVLM resolution capabilities up to 4K HD (3840 x 1600) and beyond. Concurrently, considering t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.06512","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-09T17:59:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"07f648b459094d68b3d00e94cbe3e26dcfb4798b1ebcc89dfd2bb160e9412330","abstract_canon_sha256":"8e7348888d8e03fe2d813a2ab3518c3c670ffb3c533514f10e25275f03510af5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:06:09.844994Z","signature_b64":"JdKZM8hL2/1gITHk0sxNefjF8ibP7Uv1/qzh8Df2c7iK8jMlOfXE4M+dvRAjFeABTW2CnkzdGH0KMhD+ukDQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"219da1a3b6f37db702aec84aee174ca50cc73fde560ddca53bfec0efd38d3ae9","last_reissued_at":"2026-07-05T08:06:09.844584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:06:09.844584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Wang, Conghui He, Dahua Lin, Hang Yan, Haodong Duan, Jiaqi Wang, Jifeng Dai, Jingwen Li, Kai Chen, Linke Ouyang, Pan Zhang, Songyang Zhang, Wei Li, Wenhai Wang, Wenwei Zhang, Xiaoyi Dong, Xingcheng Zhang, Xinyue Zhang, Yang Gao, Yining Li, Yuhang Cao, Yuhang Zang, Yu Qiao, Zhe Chen","submitted_at":"2024-04-09T17:59:32Z","abstract_excerpt":"The Large Vision-Language Model (LVLM) field has seen significant advancements, yet its progression has been hindered by challenges in comprehending fine-grained visual content due to limited resolution. Recent efforts have aimed to enhance the high-resolution understanding capabilities of LVLMs, yet they remain capped at approximately 1500 x 1500 pixels and constrained to a relatively narrow resolution range. This paper represents InternLM-XComposer2-4KHD, a groundbreaking exploration into elevating LVLM resolution capabilities up to 4K HD (3840 x 1600) and beyond. Concurrently, considering t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.06512","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.06512/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.06512","created_at":"2026-07-05T08:06:09.844639+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.06512v1","created_at":"2026-07-05T08:06:09.844639+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.06512","created_at":"2026-07-05T08:06:09.844639+00:00"},{"alias_kind":"pith_short_12","alias_value":"EGO2DI5W6N63","created_at":"2026-07-05T08:06:09.844639+00:00"},{"alias_kind":"pith_short_16","alias_value":"EGO2DI5W6N63OAVO","created_at":"2026-07-05T08:06:09.844639+00:00"},{"alias_kind":"pith_short_8","alias_value":"EGO2DI5W","created_at":"2026-07-05T08:06:09.844639+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22013","citing_title":"Load Testing for Machine Learning Model Serving Systems at Scale","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11853","citing_title":"Task-Aware Structured Memory for Dynamic Multi-modal In-Context Learning","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05970","citing_title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2407.12772","citing_title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2409.17146","citing_title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2407.07726","citing_title":"PaliGemma: A versatile 3B VLM for transfer","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01800","citing_title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14219","citing_title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU","json":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU.json","graph_json":"https://pith.science/api/pith-number/EGO2DI5W6N63OAVOZBFO4F2MUU/graph.json","events_json":"https://pith.science/api/pith-number/EGO2DI5W6N63OAVOZBFO4F2MUU/events.json","paper":"https://pith.science/paper/EGO2DI5W"},"agent_actions":{"view_html":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU","download_json":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU.json","view_paper":"https://pith.science/paper/EGO2DI5W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.06512&json=true","fetch_graph":"https://pith.science/api/pith-number/EGO2DI5W6N63OAVOZBFO4F2MUU/graph.json","fetch_events":"https://pith.science/api/pith-number/EGO2DI5W6N63OAVOZBFO4F2MUU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU/action/storage_attestation","attest_author":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU/action/author_attestation","sign_citation":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU/action/citation_signature","submit_replication":"https://pith.science/pith/EGO2DI5W6N63OAVOZBFO4F2MUU/action/replication_record"}},"created_at":"2026-07-05T08:06:09.844639+00:00","updated_at":"2026-07-05T08:06:09.844639+00:00"}