{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OYB3TXHMS3OATEW5BBJ3V263D3","short_pith_number":"pith:OYB3TXHM","schema_version":"1.0","canonical_sha256":"7603b9dcec96dc0992dd0853baebdb1eeb536d55fba66e30e8237529ac7b73d1","source":{"kind":"arxiv","id":"2406.08487","version":3},"attestation_state":"computed","paper":{"title":"Beyond LLaVA-HD: Diving into High-Resolution Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaoyou Fu, Liang Wang, Qingsong Wen, Rong Jin, Xue Wang, Yi-Fan Zhang, Zhang Zhang","submitted_at":"2024-06-12T17:59:49Z","abstract_excerpt":"Seeing clearly with high resolution is a foundation of Large Multimodal Models (LMMs), which has been proven to be vital for visual perception and reasoning. Existing works usually employ a straightforward resolution upscaling method, where the image consists of global and local branches, with the latter being the sliced image patches but resized to the same resolution as the former. This means that higher resolution requires more local patches, resulting in exorbitant computational expenses, and meanwhile, the dominance of local image tokens may diminish the global context. In this paper, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08487","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-12T17:59:49Z","cross_cats_sorted":[],"title_canon_sha256":"50c184a2fa163689e562941c757d5658e6b07e9b62b574564091b078726b6350","abstract_canon_sha256":"b790e72b3cbbfe0da01caec05d6bc2872d10c4aca2f69104cf9506182f5a6518"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:53.261347Z","signature_b64":"4jZ4bZXTIKflC45BM75oqzfXgDWvQznX2057yMMqXiGv2AtIAmj04OHfhyL7X7BkRfSP9LcaRV3vYBw9qnflBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7603b9dcec96dc0992dd0853baebdb1eeb536d55fba66e30e8237529ac7b73d1","last_reissued_at":"2026-07-05T08:31:53.260836Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:53.260836Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond LLaVA-HD: Diving into High-Resolution Large Multimodal Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaoyou Fu, Liang Wang, Qingsong Wen, Rong Jin, Xue Wang, Yi-Fan Zhang, Zhang Zhang","submitted_at":"2024-06-12T17:59:49Z","abstract_excerpt":"Seeing clearly with high resolution is a foundation of Large Multimodal Models (LMMs), which has been proven to be vital for visual perception and reasoning. Existing works usually employ a straightforward resolution upscaling method, where the image consists of global and local branches, with the latter being the sliced image patches but resized to the same resolution as the former. This means that higher resolution requires more local patches, resulting in exorbitant computational expenses, and meanwhile, the dominance of local image tokens may diminish the global context. In this paper, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08487","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08487/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08487","created_at":"2026-07-05T08:31:53.260898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08487v3","created_at":"2026-07-05T08:31:53.260898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08487","created_at":"2026-07-05T08:31:53.260898+00:00"},{"alias_kind":"pith_short_12","alias_value":"OYB3TXHMS3OA","created_at":"2026-07-05T08:31:53.260898+00:00"},{"alias_kind":"pith_short_16","alias_value":"OYB3TXHMS3OATEW5","created_at":"2026-07-05T08:31:53.260898+00:00"},{"alias_kind":"pith_short_8","alias_value":"OYB3TXHM","created_at":"2026-07-05T08:31:53.260898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":235,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05920","citing_title":"High-Resolution Visual Reasoning via Multi-Turn Grounding-Based Reinforcement Learning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22102","citing_title":"Mitigating Coordinate Prediction Bias from Positional Encoding Failures","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2501.01957","citing_title":"VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05126","citing_title":"ConsisVLA-4D: Advancing Spatiotemporal Consistency in Efficient 3D-Perception and 4D-Reasoning for Robotic Manipulation","ref_index":87,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3","json":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3.json","graph_json":"https://pith.science/api/pith-number/OYB3TXHMS3OATEW5BBJ3V263D3/graph.json","events_json":"https://pith.science/api/pith-number/OYB3TXHMS3OATEW5BBJ3V263D3/events.json","paper":"https://pith.science/paper/OYB3TXHM"},"agent_actions":{"view_html":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3","download_json":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3.json","view_paper":"https://pith.science/paper/OYB3TXHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08487&json=true","fetch_graph":"https://pith.science/api/pith-number/OYB3TXHMS3OATEW5BBJ3V263D3/graph.json","fetch_events":"https://pith.science/api/pith-number/OYB3TXHMS3OATEW5BBJ3V263D3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3/action/storage_attestation","attest_author":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3/action/author_attestation","sign_citation":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3/action/citation_signature","submit_replication":"https://pith.science/pith/OYB3TXHMS3OATEW5BBJ3V263D3/action/replication_record"}},"created_at":"2026-07-05T08:31:53.260898+00:00","updated_at":"2026-07-05T08:31:53.260898+00:00"}