{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:NCVZFAUTYIX76W75D4BODUNGND","short_pith_number":"pith:NCVZFAUT","schema_version":"1.0","canonical_sha256":"68ab928293c22fff5bfd1f02e1d1a668e6ba22b963fc35ad521cc7ccc031c400","source":{"kind":"arxiv","id":"1911.10683","version":5},"attestation_state":"computed","paper":{"title":"Image-based table recognition: data, model, and evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Antonio Jimeno Yepes, Elaheh ShafieiBavani, Xu Zhong","submitted_at":"2019-11-25T03:25:03Z","abstract_excerpt":"Important information that relates to a specific topic in a document is often organized in tabular format to assist readers with information retrieval and comparison, which may be difficult to provide in natural language. However, tabular data in unstructured digital documents, e.g., Portable Document Format (PDF) and images, are difficult to parse into structured machine-readable format, due to complexity and diversity in their structure and style. To facilitate image-based table recognition with deep learning, we develop the largest publicly available table recognition dataset PubTabNet (htt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.10683","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-11-25T03:25:03Z","cross_cats_sorted":[],"title_canon_sha256":"fb7325550ed5421b5c7daeb5d0b0bed2243e4f7e23f189ba75b1de24558f3c11","abstract_canon_sha256":"c3d15c1e063141839bf33634424f1d2df5c2494f81ab9060596e13d8e69a1ab1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:45:34.620372Z","signature_b64":"kjwcfCFf5oDvfk45QDIqClTYap/mm31+ruYYUuC3SwR/AWitjvQj6cjbJSLZgx9GPlogjNIN4lRXlzk8jenLAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"68ab928293c22fff5bfd1f02e1d1a668e6ba22b963fc35ad521cc7ccc031c400","last_reissued_at":"2026-07-05T00:45:34.619708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:45:34.619708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Image-based table recognition: data, model, and evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Antonio Jimeno Yepes, Elaheh ShafieiBavani, Xu Zhong","submitted_at":"2019-11-25T03:25:03Z","abstract_excerpt":"Important information that relates to a specific topic in a document is often organized in tabular format to assist readers with information retrieval and comparison, which may be difficult to provide in natural language. However, tabular data in unstructured digital documents, e.g., Portable Document Format (PDF) and images, are difficult to parse into structured machine-readable format, due to complexity and diversity in their structure and style. To facilitate image-based table recognition with deep learning, we develop the largest publicly available table recognition dataset PubTabNet (htt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.10683","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.10683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.10683","created_at":"2026-07-05T00:45:34.619782+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.10683v5","created_at":"2026-07-05T00:45:34.619782+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.10683","created_at":"2026-07-05T00:45:34.619782+00:00"},{"alias_kind":"pith_short_12","alias_value":"NCVZFAUTYIX7","created_at":"2026-07-05T00:45:34.619782+00:00"},{"alias_kind":"pith_short_16","alias_value":"NCVZFAUTYIX76W75","created_at":"2026-07-05T00:45:34.619782+00:00"},{"alias_kind":"pith_short_8","alias_value":"NCVZFAUT","created_at":"2026-07-05T00:45:34.619782+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.07534","citing_title":"PulseBench-Tab: A Multilingual Benchmark for Table Extraction with Graph-Based Evaluation","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21005","citing_title":"Building Agent Harnesses for Scientific Curation from Multimodal Sources","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27978","citing_title":"ABot-OCR Technical Report","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16861","citing_title":"Prefix-Adaptive Block Diffusion for Efficient Document Recognition","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00321","citing_title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09677","citing_title":"Logics-Parsing-Omni Technical Report","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2603.13224","citing_title":"Visual-ERM: Reward Modeling for Visual Equivalence","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23885","citing_title":"Towards Real-World Document Parsing via Realistic Scene Synthesis and Document-Aware Training","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND","json":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND.json","graph_json":"https://pith.science/api/pith-number/NCVZFAUTYIX76W75D4BODUNGND/graph.json","events_json":"https://pith.science/api/pith-number/NCVZFAUTYIX76W75D4BODUNGND/events.json","paper":"https://pith.science/paper/NCVZFAUT"},"agent_actions":{"view_html":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND","download_json":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND.json","view_paper":"https://pith.science/paper/NCVZFAUT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.10683&json=true","fetch_graph":"https://pith.science/api/pith-number/NCVZFAUTYIX76W75D4BODUNGND/graph.json","fetch_events":"https://pith.science/api/pith-number/NCVZFAUTYIX76W75D4BODUNGND/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND/action/storage_attestation","attest_author":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND/action/author_attestation","sign_citation":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND/action/citation_signature","submit_replication":"https://pith.science/pith/NCVZFAUTYIX76W75D4BODUNGND/action/replication_record"}},"created_at":"2026-07-05T00:45:34.619782+00:00","updated_at":"2026-07-05T00:45:34.619782+00:00"}