{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7C62S7VTFZM2IXPGYUVWPUXIVA","short_pith_number":"pith:7C62S7VT","schema_version":"1.0","canonical_sha256":"f8bda97eb32e59a45de6c52b67d2e8a81185f19e188c8b393f72ec1164f0ccbd","source":{"kind":"arxiv","id":"2306.08937","version":3},"attestation_state":"computed","paper":{"title":"DocumentNet: Bridging the Data Gap in Document Pre-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Alexander G. Hauptmann, Hanjun Dai, Jiayi Chen, Jin Miao, Lijun Yu, Wei Wei, Xiaoyu Sun","submitted_at":"2023-06-15T08:21:15Z","abstract_excerpt":"Document understanding tasks, in particular, Visually-rich Document Entity Retrieval (VDER), have gained significant attention in recent years thanks to their broad applications in enterprise AI. However, publicly available data have been scarce for these tasks due to strict privacy constraints and high annotation costs. To make things worse, the non-overlapping entity spaces from different datasets hinder the knowledge transfer between document types. In this paper, we propose a method to collect massive-scale and weakly labeled data from the web to benefit the training of VDER models. The co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.08937","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-06-15T08:21:15Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"40fee562538c6c01dc49c76d844f67a90b20035e9aa0804436ce71b9563f0976","abstract_canon_sha256":"cada5e7b58535911a92925444657a221bb7b2c6222f41e65d81c7ce2de3614f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:31.400400Z","signature_b64":"3J/Q7qLeGprILOfj12Z1ellx5+YRYWyzu58WsSgKPKRWkd20XDbadZGwK4q+YCl8myUFm6WRDuhRssr6uMZVCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8bda97eb32e59a45de6c52b67d2e8a81185f19e188c8b393f72ec1164f0ccbd","last_reissued_at":"2026-07-05T07:05:31.399870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:31.399870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DocumentNet: Bridging the Data Gap in Document Pre-Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Alexander G. Hauptmann, Hanjun Dai, Jiayi Chen, Jin Miao, Lijun Yu, Wei Wei, Xiaoyu Sun","submitted_at":"2023-06-15T08:21:15Z","abstract_excerpt":"Document understanding tasks, in particular, Visually-rich Document Entity Retrieval (VDER), have gained significant attention in recent years thanks to their broad applications in enterprise AI. However, publicly available data have been scarce for these tasks due to strict privacy constraints and high annotation costs. To make things worse, the non-overlapping entity spaces from different datasets hinder the knowledge transfer between document types. In this paper, we propose a method to collect massive-scale and weakly labeled data from the web to benefit the training of VDER models. The co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.08937","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.08937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.08937","created_at":"2026-07-05T07:05:31.399919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.08937v3","created_at":"2026-07-05T07:05:31.399919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.08937","created_at":"2026-07-05T07:05:31.399919+00:00"},{"alias_kind":"pith_short_12","alias_value":"7C62S7VTFZM2","created_at":"2026-07-05T07:05:31.399919+00:00"},{"alias_kind":"pith_short_16","alias_value":"7C62S7VTFZM2IXPG","created_at":"2026-07-05T07:05:31.399919+00:00"},{"alias_kind":"pith_short_8","alias_value":"7C62S7VT","created_at":"2026-07-05T07:05:31.399919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA","json":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA.json","graph_json":"https://pith.science/api/pith-number/7C62S7VTFZM2IXPGYUVWPUXIVA/graph.json","events_json":"https://pith.science/api/pith-number/7C62S7VTFZM2IXPGYUVWPUXIVA/events.json","paper":"https://pith.science/paper/7C62S7VT"},"agent_actions":{"view_html":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA","download_json":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA.json","view_paper":"https://pith.science/paper/7C62S7VT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.08937&json=true","fetch_graph":"https://pith.science/api/pith-number/7C62S7VTFZM2IXPGYUVWPUXIVA/graph.json","fetch_events":"https://pith.science/api/pith-number/7C62S7VTFZM2IXPGYUVWPUXIVA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA/action/storage_attestation","attest_author":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA/action/author_attestation","sign_citation":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA/action/citation_signature","submit_replication":"https://pith.science/pith/7C62S7VTFZM2IXPGYUVWPUXIVA/action/replication_record"}},"created_at":"2026-07-05T07:05:31.399919+00:00","updated_at":"2026-07-05T07:05:31.399919+00:00"}