{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K6MHR2YGVIJ7FSQFVMB7FBDQZ3","short_pith_number":"pith:K6MHR2YG","schema_version":"1.0","canonical_sha256":"579878eb06aa13f2ca05ab03f28470cef0681fdc1702e970f905162bb9c758d1","source":{"kind":"arxiv","id":"2309.11419","version":2},"attestation_state":"computed","paper":{"title":"KOSMOS-2.5: A Multimodal Literate Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Cha Zhang, Furu Wei, Guoxin Wang, Jingye Chen, Lei Cui, Li Dong, Shaohan Huang, Shaoxiang Wu, Shuming Ma, Tengchao Lv, Weiyao Luo, Wenhui Wang, Yaoyao Chang, Yilin Jia, Yupan Huang, Yuzhong Zhao","submitted_at":"2023-09-20T15:50:08Z","abstract_excerpt":"The automatic reading of text-intensive images represents a significant advancement toward achieving Artificial General Intelligence (AGI). In this paper we present KOSMOS-2.5, a multimodal literate model for machine reading of text-intensive images. Pre-trained on a large-scale corpus of text-intensive images, KOSMOS-2.5 excels in two distinct yet complementary transcription tasks: (1) generating spatially-aware text blocks, where each block of text is assigned spatial coordinates within the image, and (2) producing structured text output that captures both style and structure in markdown for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.11419","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-20T15:50:08Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"485f368a1332e176d880a286a97950896fca232e698e9fc87884487507f0a335","abstract_canon_sha256":"fab71a5db0f36c757aa19c3ec3d0d6bd10ba71a34ab580caa8bcedeee099006c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:34.052440Z","signature_b64":"nVwskUJI/2LcsU+ao91L+HlvyA0oIG2x4VmDQtA8JnbtosFUTgWUeDmek9S+ATPp4Li2ZTKhYxMVQ1pCKgOWDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"579878eb06aa13f2ca05ab03f28470cef0681fdc1702e970f905162bb9c758d1","last_reissued_at":"2026-07-05T08:57:34.051946Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:34.051946Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KOSMOS-2.5: A Multimodal Literate Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Cha Zhang, Furu Wei, Guoxin Wang, Jingye Chen, Lei Cui, Li Dong, Shaohan Huang, Shaoxiang Wu, Shuming Ma, Tengchao Lv, Weiyao Luo, Wenhui Wang, Yaoyao Chang, Yilin Jia, Yupan Huang, Yuzhong Zhao","submitted_at":"2023-09-20T15:50:08Z","abstract_excerpt":"The automatic reading of text-intensive images represents a significant advancement toward achieving Artificial General Intelligence (AGI). In this paper we present KOSMOS-2.5, a multimodal literate model for machine reading of text-intensive images. Pre-trained on a large-scale corpus of text-intensive images, KOSMOS-2.5 excels in two distinct yet complementary transcription tasks: (1) generating spatially-aware text blocks, where each block of text is assigned spatial coordinates within the image, and (2) producing structured text output that captures both style and structure in markdown for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.11419","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.11419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.11419","created_at":"2026-07-05T08:57:34.052001+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.11419v2","created_at":"2026-07-05T08:57:34.052001+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.11419","created_at":"2026-07-05T08:57:34.052001+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6MHR2YGVIJ7","created_at":"2026-07-05T08:57:34.052001+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6MHR2YGVIJ7FSQF","created_at":"2026-07-05T08:57:34.052001+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6MHR2YG","created_at":"2026-07-05T08:57:34.052001+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08029","citing_title":"Rethinking Small VLM Quantization: From Component-Wise Analysis to Hardware-Aware Edge Deployment","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09132","citing_title":"Vision Language Model Helps Private Information De-Identification in Vision Data","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00596","citing_title":"Semantic-Guided Reading Order Reconstruction in Historical Armenian Newspapers with LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03401","citing_title":"Towards Characterizing Scientific Image Utility and Upgradability","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29378","citing_title":"Cross-Temporal Sinhala OCR: Page-Level Adaptation and Diachronic Analysis","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09861","citing_title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18839","citing_title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19790","citing_title":"From Plausibility to Verifiability: Risk-Controlled Generative OCR with Vision-Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03903","citing_title":"CC-OCR V2: Benchmarking Large Multimodal Models for Literacy in Real-world Document Processing","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23813","citing_title":"ShredBench: Evaluating the Semantic Reasoning Capabilities of Multimodal LLMs in Document Reconstruction","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3","json":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3.json","graph_json":"https://pith.science/api/pith-number/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/graph.json","events_json":"https://pith.science/api/pith-number/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/events.json","paper":"https://pith.science/paper/K6MHR2YG"},"agent_actions":{"view_html":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3","download_json":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3.json","view_paper":"https://pith.science/paper/K6MHR2YG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.11419&json=true","fetch_graph":"https://pith.science/api/pith-number/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/graph.json","fetch_events":"https://pith.science/api/pith-number/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/action/storage_attestation","attest_author":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/action/author_attestation","sign_citation":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/action/citation_signature","submit_replication":"https://pith.science/pith/K6MHR2YGVIJ7FSQFVMB7FBDQZ3/action/replication_record"}},"created_at":"2026-07-05T08:57:34.052001+00:00","updated_at":"2026-07-05T08:57:34.052001+00:00"}