{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QPRI6YKWUDU5PUXLR3NXV6LSQM","short_pith_number":"pith:QPRI6YKW","schema_version":"1.0","canonical_sha256":"83e28f6156a0e9d7d2eb8edb7af97283290f448a4c3239878e26c37db395faf8","source":{"kind":"arxiv","id":"2403.04473","version":2},"attestation_state":"computed","paper":{"title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Biao Yang, Qiang Liu, Shuo Zhang, Xiang Bai, Yuliang Liu, Zhang Li, Zhiyin Ma","submitted_at":"2024-03-07T13:16:24Z","abstract_excerpt":"We present TextMonkey, a large multimodal model (LMM) tailored for text-centric tasks. Our approach introduces enhancement across several dimensions: By adopting Shifted Window Attention with zero-initialization, we achieve cross-window connectivity at higher input resolutions and stabilize early training; We hypothesize that images may contain redundant tokens, and by using similarity to filter out significant tokens, we can not only streamline the token length but also enhance the model's performance. Moreover, by expanding our model's capabilities to encompass text spotting and grounding, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.04473","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-07T13:16:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"55e0b703539e70340065158f8f0d539eaf55bb0743504df2aa50b95eeb0c9d7d","abstract_canon_sha256":"7420788aa91a20b89b009739bf2437eb6c1a10661b1e00124e6b43199a23019a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:21.813122Z","signature_b64":"u/gqvznrHKBZ0mt+kHnOnmo6KCHdV9pmamoMEvOeL5RsHG4Fs0Cj5IKUB51MvAHmc4bg+31C4lPMobZvZkDTAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83e28f6156a0e9d7d2eb8edb7af97283290f448a4c3239878e26c37db395faf8","last_reissued_at":"2026-07-05T07:56:21.812659Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:21.812659Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Biao Yang, Qiang Liu, Shuo Zhang, Xiang Bai, Yuliang Liu, Zhang Li, Zhiyin Ma","submitted_at":"2024-03-07T13:16:24Z","abstract_excerpt":"We present TextMonkey, a large multimodal model (LMM) tailored for text-centric tasks. Our approach introduces enhancement across several dimensions: By adopting Shifted Window Attention with zero-initialization, we achieve cross-window connectivity at higher input resolutions and stabilize early training; We hypothesize that images may contain redundant tokens, and by using similarity to filter out significant tokens, we can not only streamline the token length but also enhance the model's performance. Moreover, by expanding our model's capabilities to encompass text spotting and grounding, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.04473","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.04473/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.04473","created_at":"2026-07-05T07:56:21.812719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.04473v2","created_at":"2026-07-05T07:56:21.812719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.04473","created_at":"2026-07-05T07:56:21.812719+00:00"},{"alias_kind":"pith_short_12","alias_value":"QPRI6YKWUDU5","created_at":"2026-07-05T07:56:21.812719+00:00"},{"alias_kind":"pith_short_16","alias_value":"QPRI6YKWUDU5PUXL","created_at":"2026-07-05T07:56:21.812719+00:00"},{"alias_kind":"pith_short_8","alias_value":"QPRI6YKW","created_at":"2026-07-05T07:56:21.812719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.05970","citing_title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.21169","citing_title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09861","citing_title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01704","citing_title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22186","citing_title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2407.03320","citing_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10594","citing_title":"VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12937","citing_title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18472","citing_title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00161","citing_title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16821","citing_title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12812","citing_title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12812","citing_title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07419","citing_title":"ReAlign: Optimizing the Visual Document Retriever with Reasoning-Guided Fine-Grained Alignment","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01800","citing_title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM","json":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM.json","graph_json":"https://pith.science/api/pith-number/QPRI6YKWUDU5PUXLR3NXV6LSQM/graph.json","events_json":"https://pith.science/api/pith-number/QPRI6YKWUDU5PUXLR3NXV6LSQM/events.json","paper":"https://pith.science/paper/QPRI6YKW"},"agent_actions":{"view_html":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM","download_json":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM.json","view_paper":"https://pith.science/paper/QPRI6YKW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.04473&json=true","fetch_graph":"https://pith.science/api/pith-number/QPRI6YKWUDU5PUXLR3NXV6LSQM/graph.json","fetch_events":"https://pith.science/api/pith-number/QPRI6YKWUDU5PUXLR3NXV6LSQM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM/action/storage_attestation","attest_author":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM/action/author_attestation","sign_citation":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM/action/citation_signature","submit_replication":"https://pith.science/pith/QPRI6YKWUDU5PUXLR3NXV6LSQM/action/replication_record"}},"created_at":"2026-07-05T07:56:21.812719+00:00","updated_at":"2026-07-05T07:56:21.812719+00:00"}