{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZGW7J2Y5SWB2NJH6FESPD7HQO7","short_pith_number":"pith:ZGW7J2Y5","schema_version":"1.0","canonical_sha256":"c9adf4eb1d9583a6a4fe2924f1fcf077e07e17f5ad59ea3030e16a31bcaae72e","source":{"kind":"arxiv","id":"2412.02210","version":3},"attestation_state":"computed","paper":{"title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Humen Zhong, Jianqiang Wan, Jun Tang, Junyang Lin, Lianwen Jin, Mingkun Yang, Pengfei Wang, Peng Wang, Shuai Bai, Xuejing Liu, Zhaohai Li, Zhibo Yang","submitted_at":"2024-12-03T07:03:25Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated impressive performance in recognizing document images with natural language instructions. However, it remains unclear to what extent capabilities in literacy with rich structure and fine-grained visual challenges. The current landscape lacks a comprehensive benchmark to effectively measure the literate capabilities of LMMs. Existing benchmarks are often limited by narrow scenarios and specified tasks. To this end, we introduce CC-OCR, a comprehensive benchmark that possesses a diverse range of scenarios, tasks, and challenges. CC-OCR comprises f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.02210","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-03T07:03:25Z","cross_cats_sorted":[],"title_canon_sha256":"64e3fd77dad6a9f8a1f6f200dd329b7f27e028dec2e0d9de6e83cb68fcb97eae","abstract_canon_sha256":"0375618fe87a2fe58e5071d20265da0b36a32c1817fa54f41d1509cac36f5760"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:46:55.639570Z","signature_b64":"9dbOeMph2HJ9SkQc1HJl0btqiVQiHNQKHz3OCfhpcVPVAvQt4YogxVgC8V40VW0YYQcMtz7U0IqoZxnugE6UDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9adf4eb1d9583a6a4fe2924f1fcf077e07e17f5ad59ea3030e16a31bcaae72e","last_reissued_at":"2026-07-05T09:46:55.638985Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:46:55.638985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Humen Zhong, Jianqiang Wan, Jun Tang, Junyang Lin, Lianwen Jin, Mingkun Yang, Pengfei Wang, Peng Wang, Shuai Bai, Xuejing Liu, Zhaohai Li, Zhibo Yang","submitted_at":"2024-12-03T07:03:25Z","abstract_excerpt":"Large Multimodal Models (LMMs) have demonstrated impressive performance in recognizing document images with natural language instructions. However, it remains unclear to what extent capabilities in literacy with rich structure and fine-grained visual challenges. The current landscape lacks a comprehensive benchmark to effectively measure the literate capabilities of LMMs. Existing benchmarks are often limited by narrow scenarios and specified tasks. To this end, we introduce CC-OCR, a comprehensive benchmark that possesses a diverse range of scenarios, tasks, and challenges. CC-OCR comprises f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.02210","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.02210/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.02210","created_at":"2026-07-05T09:46:55.639065+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.02210v3","created_at":"2026-07-05T09:46:55.639065+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.02210","created_at":"2026-07-05T09:46:55.639065+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZGW7J2Y5SWB2","created_at":"2026-07-05T09:46:55.639065+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZGW7J2Y5SWB2NJH6","created_at":"2026-07-05T09:46:55.639065+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZGW7J2Y5","created_at":"2026-07-05T09:46:55.639065+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11477","citing_title":"Towards Fully Automated Exam Grading: Fairness-Aware Recognition of Handwritten Answers with Foundation Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26712","citing_title":"METATR: A Multilingual, Evolving Benchmark for Automatic Text Recognition","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10016","citing_title":"Training-Free Multimodal Large Language Model Orchestration","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10016","citing_title":"Training-Free Multimodal Large Language Model Orchestration","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14998","citing_title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22186","citing_title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23885","citing_title":"Towards Real-World Document Parsing via Realistic Scene Synthesis and Document-Aware Training","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11301","citing_title":"LatentRouter: Can We Choose the Right Multimodal Model Before Seeing Its Answer?","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7","json":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7.json","graph_json":"https://pith.science/api/pith-number/ZGW7J2Y5SWB2NJH6FESPD7HQO7/graph.json","events_json":"https://pith.science/api/pith-number/ZGW7J2Y5SWB2NJH6FESPD7HQO7/events.json","paper":"https://pith.science/paper/ZGW7J2Y5"},"agent_actions":{"view_html":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7","download_json":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7.json","view_paper":"https://pith.science/paper/ZGW7J2Y5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.02210&json=true","fetch_graph":"https://pith.science/api/pith-number/ZGW7J2Y5SWB2NJH6FESPD7HQO7/graph.json","fetch_events":"https://pith.science/api/pith-number/ZGW7J2Y5SWB2NJH6FESPD7HQO7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7/action/storage_attestation","attest_author":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7/action/author_attestation","sign_citation":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7/action/citation_signature","submit_replication":"https://pith.science/pith/ZGW7J2Y5SWB2NJH6FESPD7HQO7/action/replication_record"}},"created_at":"2026-07-05T09:46:55.639065+00:00","updated_at":"2026-07-05T09:46:55.639065+00:00"}