{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:UGJP7HMPGG2KJZ266MGPLET2GV","short_pith_number":"pith:UGJP7HMP","schema_version":"1.0","canonical_sha256":"a192ff9d8f31b4a4e75ef30cf5927a35755683da124e642292a490664a6ab7da","source":{"kind":"arxiv","id":"2607.16203","version":1},"attestation_state":"computed","paper":{"title":"DocOCR-Eval: A Correction-Based Framework for OCR Tool Selection Without Ground Truth","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Lawrence Chun Man Lau, Puzhen Wu, Sirui Li, Wei Liu, Yifan Peng, Yihao Ding, Zihan Xu","submitted_at":"2026-05-05T10:59:34Z","abstract_excerpt":"Document parsing is a foundational step for document understanding tasks such as visual question answering and key information extraction, as it transforms unstructured scanned images into structured representations by extracting textual, visual, and layout information. While numerous Optical Character Recognition (OCR) engines and multimodal large language models (MLLMs) have been developed for this purpose, selecting an appropriate document parsing solution for a given document collection remains challenging, particularly in label-scarce settings. In this work, we conduct a systematic evalua"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.16203","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-05-05T10:59:34Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"dc65cb284ba57d5d8a1b4acdbe45500c5c5f3e5adc0abd874882a0787add62bb","abstract_canon_sha256":"cf770872ccc74b715525e0935d7549a863afb6f0c6cf7ac7ddc605596660a839"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T00:20:05.550507Z","signature_b64":"Q4b3pWARZB3/kZlAQ5awBwLUYsGQRjg/QBvyCsHgPc/HPqsrLGKYzT/tMP+gaRgDcpmG1KbslhaNsTJyNzI9Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a192ff9d8f31b4a4e75ef30cf5927a35755683da124e642292a490664a6ab7da","last_reissued_at":"2026-07-21T00:20:05.549541Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T00:20:05.549541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DocOCR-Eval: A Correction-Based Framework for OCR Tool Selection Without Ground Truth","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Lawrence Chun Man Lau, Puzhen Wu, Sirui Li, Wei Liu, Yifan Peng, Yihao Ding, Zihan Xu","submitted_at":"2026-05-05T10:59:34Z","abstract_excerpt":"Document parsing is a foundational step for document understanding tasks such as visual question answering and key information extraction, as it transforms unstructured scanned images into structured representations by extracting textual, visual, and layout information. While numerous Optical Character Recognition (OCR) engines and multimodal large language models (MLLMs) have been developed for this purpose, selecting an appropriate document parsing solution for a given document collection remains challenging, particularly in label-scarce settings. In this work, we conduct a systematic evalua"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.16203","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.16203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.16203","created_at":"2026-07-21T00:20:05.550005+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.16203v1","created_at":"2026-07-21T00:20:05.550005+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.16203","created_at":"2026-07-21T00:20:05.550005+00:00"},{"alias_kind":"pith_short_12","alias_value":"UGJP7HMPGG2K","created_at":"2026-07-21T00:20:05.550005+00:00"},{"alias_kind":"pith_short_16","alias_value":"UGJP7HMPGG2KJZ26","created_at":"2026-07-21T00:20:05.550005+00:00"},{"alias_kind":"pith_short_8","alias_value":"UGJP7HMP","created_at":"2026-07-21T00:20:05.550005+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV","json":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV.json","graph_json":"https://pith.science/api/pith-number/UGJP7HMPGG2KJZ266MGPLET2GV/graph.json","events_json":"https://pith.science/api/pith-number/UGJP7HMPGG2KJZ266MGPLET2GV/events.json","paper":"https://pith.science/paper/UGJP7HMP"},"agent_actions":{"view_html":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV","download_json":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV.json","view_paper":"https://pith.science/paper/UGJP7HMP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.16203&json=true","fetch_graph":"https://pith.science/api/pith-number/UGJP7HMPGG2KJZ266MGPLET2GV/graph.json","fetch_events":"https://pith.science/api/pith-number/UGJP7HMPGG2KJZ266MGPLET2GV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV/action/storage_attestation","attest_author":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV/action/author_attestation","sign_citation":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV/action/citation_signature","submit_replication":"https://pith.science/pith/UGJP7HMPGG2KJZ266MGPLET2GV/action/replication_record"}},"created_at":"2026-07-21T00:20:05.550005+00:00","updated_at":"2026-07-21T00:20:05.550005+00:00"}