{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XJCMGDXQOBQI72GR2346JL5DOX","short_pith_number":"pith:XJCMGDXQ","schema_version":"1.0","canonical_sha256":"ba44c30ef070608fe8d1d6f9e4afa375cfe3a914c9f0156af4a7b9d4abeedb74","source":{"kind":"arxiv","id":"2312.10188","version":1},"attestation_state":"computed","paper":{"title":"WordScape: a Pipeline to extract multilingual, visually rich Documents with Layout Annotations from Web Crawl Data","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anton Alexandrov, Bo Li, Carlo Siebenschuh, Ce Zhang, Georgios Tsolakis, Haris Jabbar, Ian Foster, Maurice Weber, Rick Stevens, Rory Butler, Valdemar Thanner","submitted_at":"2023-12-15T20:28:31Z","abstract_excerpt":"We introduce WordScape, a novel pipeline for the creation of cross-disciplinary, multilingual corpora comprising millions of pages with annotations for document layout detection. Relating visual and textual items on document pages has gained further significance with the advent of multimodal models. Various approaches proved effective for visual question answering or layout segmentation. However, the interplay of text, tables, and visuals remains challenging for a variety of document understanding tasks. In particular, many models fail to generalize well to diverse domains and new languages du"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.10188","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-15T20:28:31Z","cross_cats_sorted":[],"title_canon_sha256":"87b137b62c779e0d94c121b0f36d4636b34fb6411b8d26a30bc2a7e48d4b7a9f","abstract_canon_sha256":"f148867899180ca6deec1936a08cce029cde3c24582ebd6ee2095857648189ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:02.592783Z","signature_b64":"RpNzM1jOcu8BscJrRL5sWVf4LWy5xqEOYwoG2f8H4e7eZ2U+M5BpJebmKH4+4D7HL80C9RoZPnAu9kygH1eDBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba44c30ef070608fe8d1d6f9e4afa375cfe3a914c9f0156af4a7b9d4abeedb74","last_reissued_at":"2026-07-05T07:25:02.592291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:02.592291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WordScape: a Pipeline to extract multilingual, visually rich Documents with Layout Annotations from Web Crawl Data","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anton Alexandrov, Bo Li, Carlo Siebenschuh, Ce Zhang, Georgios Tsolakis, Haris Jabbar, Ian Foster, Maurice Weber, Rick Stevens, Rory Butler, Valdemar Thanner","submitted_at":"2023-12-15T20:28:31Z","abstract_excerpt":"We introduce WordScape, a novel pipeline for the creation of cross-disciplinary, multilingual corpora comprising millions of pages with annotations for document layout detection. Relating visual and textual items on document pages has gained further significance with the advent of multimodal models. Various approaches proved effective for visual question answering or layout segmentation. However, the interplay of text, tables, and visuals remains challenging for a variety of document understanding tasks. In particular, many models fail to generalize well to diverse domains and new languages du"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.10188","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.10188/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.10188","created_at":"2026-07-05T07:25:02.592359+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.10188v1","created_at":"2026-07-05T07:25:02.592359+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.10188","created_at":"2026-07-05T07:25:02.592359+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJCMGDXQOBQI","created_at":"2026-07-05T07:25:02.592359+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJCMGDXQOBQI72GR","created_at":"2026-07-05T07:25:02.592359+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJCMGDXQ","created_at":"2026-07-05T07:25:02.592359+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19866","citing_title":"Structured Layout Priors for Robust Out-of-Distribution Visual Document Understanding","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX","json":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX.json","graph_json":"https://pith.science/api/pith-number/XJCMGDXQOBQI72GR2346JL5DOX/graph.json","events_json":"https://pith.science/api/pith-number/XJCMGDXQOBQI72GR2346JL5DOX/events.json","paper":"https://pith.science/paper/XJCMGDXQ"},"agent_actions":{"view_html":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX","download_json":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX.json","view_paper":"https://pith.science/paper/XJCMGDXQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.10188&json=true","fetch_graph":"https://pith.science/api/pith-number/XJCMGDXQOBQI72GR2346JL5DOX/graph.json","fetch_events":"https://pith.science/api/pith-number/XJCMGDXQOBQI72GR2346JL5DOX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX/action/storage_attestation","attest_author":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX/action/author_attestation","sign_citation":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX/action/citation_signature","submit_replication":"https://pith.science/pith/XJCMGDXQOBQI72GR2346JL5DOX/action/replication_record"}},"created_at":"2026-07-05T07:25:02.592359+00:00","updated_at":"2026-07-05T07:25:02.592359+00:00"}