{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TW6GKHXR4SNFFGDWGIXH5ZD4NH","short_pith_number":"pith:TW6GKHXR","schema_version":"1.0","canonical_sha256":"9dbc651ef1e49a529876322e7ee47c69e87d89872ccfe5794228882a50969cf3","source":{"kind":"arxiv","id":"2312.02694","version":2},"attestation_state":"computed","paper":{"title":"UPOCR: Towards Unified Pixel-Level OCR Interface","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongyu Liu, Dezhi Peng, Fengjun Guo, Jiaxin Zhang, Kai Ding, Lianwen Jin, Yongxin Shi, Zhenhua Yang","submitted_at":"2023-12-05T11:53:17Z","abstract_excerpt":"Existing optical character recognition (OCR) methods rely on task-specific designs with divergent paradigms, architectures, and training strategies, which significantly increases the complexity of research and maintenance and hinders the fast deployment in applications. To this end, we propose UPOCR, a simple-yet-effective generalist model for Unified Pixel-level OCR interface. Specifically, the UPOCR unifies the paradigm of diverse OCR tasks as image-to-image transformation and the architecture as a vision Transformer (ViT)-based encoder-decoder with learnable task prompts. The prompts push t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02694","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-05T11:53:17Z","cross_cats_sorted":[],"title_canon_sha256":"9ecbf71874cb13ba1f1f93fc041c41c1460c8d1ca6158ddd10d128ed601baf75","abstract_canon_sha256":"084d4facff560c98c14a41197b8d28e3a73ba75ed610517182844e3213a9f600"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-27T01:04:46.693148Z","signature_b64":"N1q3lqT1eDXgaU3ZzsNUXY++tWCZxd5kcEJat+9crjursB68YImjoPfQx639tMdm6E7y/xJ5z3hvuad66EtwAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9dbc651ef1e49a529876322e7ee47c69e87d89872ccfe5794228882a50969cf3","last_reissued_at":"2026-05-27T01:04:46.692503Z","signature_status":"signed_v1","first_computed_at":"2026-05-27T01:04:46.692503Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UPOCR: Towards Unified Pixel-Level OCR Interface","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chongyu Liu, Dezhi Peng, Fengjun Guo, Jiaxin Zhang, Kai Ding, Lianwen Jin, Yongxin Shi, Zhenhua Yang","submitted_at":"2023-12-05T11:53:17Z","abstract_excerpt":"Existing optical character recognition (OCR) methods rely on task-specific designs with divergent paradigms, architectures, and training strategies, which significantly increases the complexity of research and maintenance and hinders the fast deployment in applications. To this end, we propose UPOCR, a simple-yet-effective generalist model for Unified Pixel-level OCR interface. Specifically, the UPOCR unifies the paradigm of diverse OCR tasks as image-to-image transformation and the architecture as a vision Transformer (ViT)-based encoder-decoder with learnable task prompts. The prompts push t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02694","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02694","created_at":"2026-05-27T01:04:46.692580+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02694v2","created_at":"2026-05-27T01:04:46.692580+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02694","created_at":"2026-05-27T01:04:46.692580+00:00"},{"alias_kind":"pith_short_12","alias_value":"TW6GKHXR4SNF","created_at":"2026-05-27T01:04:46.692580+00:00"},{"alias_kind":"pith_short_16","alias_value":"TW6GKHXR4SNFFGDW","created_at":"2026-05-27T01:04:46.692580+00:00"},{"alias_kind":"pith_short_8","alias_value":"TW6GKHXR","created_at":"2026-05-27T01:04:46.692580+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.15523","citing_title":"InstructOCR: Instruction Boosting Scene Text Spotting","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH","json":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH.json","graph_json":"https://pith.science/api/pith-number/TW6GKHXR4SNFFGDWGIXH5ZD4NH/graph.json","events_json":"https://pith.science/api/pith-number/TW6GKHXR4SNFFGDWGIXH5ZD4NH/events.json","paper":"https://pith.science/paper/TW6GKHXR"},"agent_actions":{"view_html":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH","download_json":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH.json","view_paper":"https://pith.science/paper/TW6GKHXR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02694&json=true","fetch_graph":"https://pith.science/api/pith-number/TW6GKHXR4SNFFGDWGIXH5ZD4NH/graph.json","fetch_events":"https://pith.science/api/pith-number/TW6GKHXR4SNFFGDWGIXH5ZD4NH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH/action/storage_attestation","attest_author":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH/action/author_attestation","sign_citation":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH/action/citation_signature","submit_replication":"https://pith.science/pith/TW6GKHXR4SNFFGDWGIXH5ZD4NH/action/replication_record"}},"created_at":"2026-05-27T01:04:46.692580+00:00","updated_at":"2026-05-27T01:04:46.692580+00:00"}