{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IYEUQ4IDKVAEUZ6VHA5J62PCDB","short_pith_number":"pith:IYEUQ4ID","schema_version":"1.0","canonical_sha256":"460948710355404a67d5383a9f69e21863846cf079e3c455156c6b74afda0448","source":{"kind":"arxiv","id":"2501.15558","version":1},"attestation_state":"computed","paper":{"title":"Ocean-OCR: Towards General OCR Application via a Vision-Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongdong Kuang, Fengyu Zhang, Jianhua Xu, Lingfeng Ming, MingAn Lin, Song Chen, Tao Zhang, Weipeng Chen, Xinyu Guo, Yadong Li, Youwei Zhang, Yuran Wang, Zenan Zhou","submitted_at":"2025-01-26T15:20:39Z","abstract_excerpt":"Multimodal large language models (MLLMs) have shown impressive capabilities across various domains, excelling in processing and understanding information from multiple modalities. Despite the rapid progress made previously, insufficient OCR ability hinders MLLMs from excelling in text-related tasks. In this paper, we present \\textbf{Ocean-OCR}, a 3B MLLM with state-of-the-art performance on various OCR scenarios and comparable understanding ability on general tasks. We employ Native Resolution ViT to enable variable resolution input and utilize a substantial collection of high-quality OCR data"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.15558","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-26T15:20:39Z","cross_cats_sorted":[],"title_canon_sha256":"a6a0b5f562386a78622ceeaad0d82a56fa7b6d1a308baf93b8e8ea52088ebcdc","abstract_canon_sha256":"25f141f3c68c4a9537a6163ba273d0b8a78f5109363f6ebd829fc888e8b8692a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:31.121493Z","signature_b64":"DC5R15omJDM4PLqtMlMo33reOUZpo613YwQ5b8IfI4W5FNk5yEl7EFxE+I4jz2USVdZlxlnLKCm8apgjYK+oCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"460948710355404a67d5383a9f69e21863846cf079e3c455156c6b74afda0448","last_reissued_at":"2026-07-05T10:05:31.120999Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:31.120999Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ocean-OCR: Towards General OCR Application via a Vision-Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongdong Kuang, Fengyu Zhang, Jianhua Xu, Lingfeng Ming, MingAn Lin, Song Chen, Tao Zhang, Weipeng Chen, Xinyu Guo, Yadong Li, Youwei Zhang, Yuran Wang, Zenan Zhou","submitted_at":"2025-01-26T15:20:39Z","abstract_excerpt":"Multimodal large language models (MLLMs) have shown impressive capabilities across various domains, excelling in processing and understanding information from multiple modalities. Despite the rapid progress made previously, insufficient OCR ability hinders MLLMs from excelling in text-related tasks. In this paper, we present \\textbf{Ocean-OCR}, a 3B MLLM with state-of-the-art performance on various OCR scenarios and comparable understanding ability on general tasks. We employ Native Resolution ViT to enable variable resolution input and utilize a substantial collection of high-quality OCR data"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.15558","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.15558/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.15558","created_at":"2026-07-05T10:05:31.121052+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.15558v1","created_at":"2026-07-05T10:05:31.121052+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.15558","created_at":"2026-07-05T10:05:31.121052+00:00"},{"alias_kind":"pith_short_12","alias_value":"IYEUQ4IDKVAE","created_at":"2026-07-05T10:05:31.121052+00:00"},{"alias_kind":"pith_short_16","alias_value":"IYEUQ4IDKVAEUZ6V","created_at":"2026-07-05T10:05:31.121052+00:00"},{"alias_kind":"pith_short_8","alias_value":"IYEUQ4ID","created_at":"2026-07-05T10:05:31.121052+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02025","citing_title":"Evaluating Vision-Language Models as a Zero-Shot Learning Alternative to You Only Look Once and Optical Character Recognition for Nigerian License Plate Recognition","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00392","citing_title":"RTPrune: Reading-Twice Inspired Token Pruning for Efficient DeepSeek-OCR Inference","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12623","citing_title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22186","citing_title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19790","citing_title":"From Plausibility to Verifiability: Risk-Controlled Generative OCR with Vision-Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12623","citing_title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00392","citing_title":"RTPrune: Reading-Twice Inspired Token Pruning for Efficient DeepSeek-OCR Inference","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00392","citing_title":"RTPrune: Reading-Twice Inspired Token Pruning for Efficient DeepSeek-OCR Inference","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04771","citing_title":"MinerU2.5-Pro: Pushing the Limits of Data-Centric Document Parsing at Scale","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13403","citing_title":"Why Multimodal In-Context Learning Lags Behind? Unveiling the Inner Mechanisms and Bottlenecks","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB","json":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB.json","graph_json":"https://pith.science/api/pith-number/IYEUQ4IDKVAEUZ6VHA5J62PCDB/graph.json","events_json":"https://pith.science/api/pith-number/IYEUQ4IDKVAEUZ6VHA5J62PCDB/events.json","paper":"https://pith.science/paper/IYEUQ4ID"},"agent_actions":{"view_html":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB","download_json":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB.json","view_paper":"https://pith.science/paper/IYEUQ4ID","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.15558&json=true","fetch_graph":"https://pith.science/api/pith-number/IYEUQ4IDKVAEUZ6VHA5J62PCDB/graph.json","fetch_events":"https://pith.science/api/pith-number/IYEUQ4IDKVAEUZ6VHA5J62PCDB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB/action/storage_attestation","attest_author":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB/action/author_attestation","sign_citation":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB/action/citation_signature","submit_replication":"https://pith.science/pith/IYEUQ4IDKVAEUZ6VHA5J62PCDB/action/replication_record"}},"created_at":"2026-07-05T10:05:31.121052+00:00","updated_at":"2026-07-05T10:05:31.121052+00:00"}