{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PLRRRHH7ZAOBTRYT4ASUCYC4PZ","short_pith_number":"pith:PLRRRHH7","schema_version":"1.0","canonical_sha256":"7ae3189cffc81c19c713e02541605c7e71a480017ceae2c4bbc5cda93477c98d","source":{"kind":"arxiv","id":"2310.12430","version":1},"attestation_state":"computed","paper":{"title":"DocXChain: A Powerful Open-Source Toolchain for Document Parsing and Beyond","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Cong Yao","submitted_at":"2023-10-19T02:49:09Z","abstract_excerpt":"In this report, we introduce DocXChain, a powerful open-source toolchain for document parsing, which is designed and developed to automatically convert the rich information embodied in unstructured documents, such as text, tables and charts, into structured representations that are readable and manipulable by machines. Specifically, basic capabilities, including text detection, text recognition, table structure recognition and layout analysis, are provided. Upon these basic capabilities, we also build a set of fully functional pipelines for document parsing, i.e., general text reading, table p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.12430","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-19T02:49:09Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a2f9b6d740d320f945593e58bc88b8eff8cedeb7ca9ad9ea526df004af2e9253","abstract_canon_sha256":"fa0fb020af6c9f746cd4ee160fc19b4a1a8235b0e441823003a5272bbcd1fde1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:37.900364Z","signature_b64":"ahVOL4Brt/WRHIAvGeM7pfmiC+p7OfwIAnkRr6dUuVf3RVeUsLx/MEStvDAvHxOr31p72geB0Cr5dmYrhqVTCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7ae3189cffc81c19c713e02541605c7e71a480017ceae2c4bbc5cda93477c98d","last_reissued_at":"2026-07-05T07:02:37.899894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:37.899894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DocXChain: A Powerful Open-Source Toolchain for Document Parsing and Beyond","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Cong Yao","submitted_at":"2023-10-19T02:49:09Z","abstract_excerpt":"In this report, we introduce DocXChain, a powerful open-source toolchain for document parsing, which is designed and developed to automatically convert the rich information embodied in unstructured documents, such as text, tables and charts, into structured representations that are readable and manipulable by machines. Specifically, basic capabilities, including text detection, text recognition, table structure recognition and layout analysis, are provided. Upon these basic capabilities, we also build a set of fully functional pipelines for document parsing, i.e., general text reading, table p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.12430","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.12430/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.12430","created_at":"2026-07-05T07:02:37.899952+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.12430v1","created_at":"2026-07-05T07:02:37.899952+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.12430","created_at":"2026-07-05T07:02:37.899952+00:00"},{"alias_kind":"pith_short_12","alias_value":"PLRRRHH7ZAOB","created_at":"2026-07-05T07:02:37.899952+00:00"},{"alias_kind":"pith_short_16","alias_value":"PLRRRHH7ZAOBTRYT","created_at":"2026-07-05T07:02:37.899952+00:00"},{"alias_kind":"pith_short_8","alias_value":"PLRRRHH7","created_at":"2026-07-05T07:02:37.899952+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.21169","citing_title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","ref_index":277,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18839","citing_title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ","json":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ.json","graph_json":"https://pith.science/api/pith-number/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/graph.json","events_json":"https://pith.science/api/pith-number/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/events.json","paper":"https://pith.science/paper/PLRRRHH7"},"agent_actions":{"view_html":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ","download_json":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ.json","view_paper":"https://pith.science/paper/PLRRRHH7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.12430&json=true","fetch_graph":"https://pith.science/api/pith-number/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/graph.json","fetch_events":"https://pith.science/api/pith-number/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/action/storage_attestation","attest_author":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/action/author_attestation","sign_citation":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/action/citation_signature","submit_replication":"https://pith.science/pith/PLRRRHH7ZAOBTRYT4ASUCYC4PZ/action/replication_record"}},"created_at":"2026-07-05T07:02:37.899952+00:00","updated_at":"2026-07-05T07:02:37.899952+00:00"}