{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UQHG5HRZX2WJF4SDE3BWUPLNVX","short_pith_number":"pith:UQHG5HRZ","schema_version":"1.0","canonical_sha256":"a40e6e9e39beac92f24326c36a3d6dadf181cb94b92afc404fb1d5fc0ec917f7","source":{"kind":"arxiv","id":"2311.04219","version":1},"attestation_state":"computed","paper":{"title":"OtterHD: A High-Resolution Multi-modality Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Li, Fanyi Pu, Jingkang Yang, Peiyuan Zhang, Yuanhan Zhang, Ziwei Liu","submitted_at":"2023-11-07T18:59:58Z","abstract_excerpt":"In this paper, we present OtterHD-8B, an innovative multimodal model evolved from Fuyu-8B, specifically engineered to interpret high-resolution visual inputs with granular precision. Unlike conventional models that are constrained by fixed-size vision encoders, OtterHD-8B boasts the ability to handle flexible input dimensions, ensuring its versatility across various inference requirements. Alongside this model, we introduce MagnifierBench, an evaluation framework designed to scrutinize models' ability to discern minute details and spatial relationships of small objects. Our comparative analysi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.04219","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-07T18:59:58Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9c5eb8652f695c155dc8383b8169c28a2bd5ebdaa8a8d78293af32e1633b2de6","abstract_canon_sha256":"da55bf7d9e626740cffd2fa0d482cc38ab82eabd66ed17c18e1d78605b843611"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:10:12.404786Z","signature_b64":"qPchaJapmICJCxyaYjgkXnEOZw4BLTIoKx870Yic0KIIIGxGHaGA5N7CiAAK2I740M277rlDkhfL6K/FUrsrDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a40e6e9e39beac92f24326c36a3d6dadf181cb94b92afc404fb1d5fc0ec917f7","last_reissued_at":"2026-07-05T07:10:12.404295Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:10:12.404295Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OtterHD: A High-Resolution Multi-modality Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Li, Fanyi Pu, Jingkang Yang, Peiyuan Zhang, Yuanhan Zhang, Ziwei Liu","submitted_at":"2023-11-07T18:59:58Z","abstract_excerpt":"In this paper, we present OtterHD-8B, an innovative multimodal model evolved from Fuyu-8B, specifically engineered to interpret high-resolution visual inputs with granular precision. Unlike conventional models that are constrained by fixed-size vision encoders, OtterHD-8B boasts the ability to handle flexible input dimensions, ensuring its versatility across various inference requirements. Alongside this model, we introduce MagnifierBench, an evaluation framework designed to scrutinize models' ability to discern minute details and spatial relationships of small objects. Our comparative analysi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.04219","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.04219/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.04219","created_at":"2026-07-05T07:10:12.404361+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.04219v1","created_at":"2026-07-05T07:10:12.404361+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.04219","created_at":"2026-07-05T07:10:12.404361+00:00"},{"alias_kind":"pith_short_12","alias_value":"UQHG5HRZX2WJ","created_at":"2026-07-05T07:10:12.404361+00:00"},{"alias_kind":"pith_short_16","alias_value":"UQHG5HRZX2WJF4SD","created_at":"2026-07-05T07:10:12.404361+00:00"},{"alias_kind":"pith_short_8","alias_value":"UQHG5HRZ","created_at":"2026-07-05T07:10:12.404361+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02089","citing_title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17030","citing_title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","ref_index":167,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":246,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13923","citing_title":"Qwen2.5-VL Technical Report","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00574","citing_title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2409.07825","citing_title":"Deep Multimodal Learning with Missing Modality: A Survey","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2403.18814","citing_title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17247","citing_title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2305.03726","citing_title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2409.17146","citing_title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08329","citing_title":"An Efficient Token Compression Framework for Visual Object Tracking","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00789","citing_title":"Make Your LVLM KV Cache More Lightweight","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX","json":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX.json","graph_json":"https://pith.science/api/pith-number/UQHG5HRZX2WJF4SDE3BWUPLNVX/graph.json","events_json":"https://pith.science/api/pith-number/UQHG5HRZX2WJF4SDE3BWUPLNVX/events.json","paper":"https://pith.science/paper/UQHG5HRZ"},"agent_actions":{"view_html":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX","download_json":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX.json","view_paper":"https://pith.science/paper/UQHG5HRZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.04219&json=true","fetch_graph":"https://pith.science/api/pith-number/UQHG5HRZX2WJF4SDE3BWUPLNVX/graph.json","fetch_events":"https://pith.science/api/pith-number/UQHG5HRZX2WJF4SDE3BWUPLNVX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX/action/storage_attestation","attest_author":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX/action/author_attestation","sign_citation":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX/action/citation_signature","submit_replication":"https://pith.science/pith/UQHG5HRZX2WJF4SDE3BWUPLNVX/action/replication_record"}},"created_at":"2026-07-05T07:10:12.404361+00:00","updated_at":"2026-07-05T07:10:12.404361+00:00"}