{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QFHJ5KZRRLEMJDD5G3OQWEAVLB","short_pith_number":"pith:QFHJ5KZR","schema_version":"1.0","canonical_sha256":"814e9eab318ac8c48c7d36dd0b1015586020abb3e651376582ed1d29b541d621","source":{"kind":"arxiv","id":"2403.01698","version":1},"attestation_state":"computed","paper":{"title":"Hypertext Entity Extraction in Webpage","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bo Shao, Daxin Jiang, Hai Zhao, Linjun Shou, Ming Gong, Tianqiao Liu, Yifei Yang","submitted_at":"2024-03-04T03:21:40Z","abstract_excerpt":"Webpage entity extraction is a fundamental natural language processing task in both research and applications. Nowadays, the majority of webpage entity extraction models are trained on structured datasets which strive to retain textual content and its structure information. However, existing datasets all overlook the rich hypertext features (e.g., font color, font size) which show their effectiveness in previous works. To this end, we first collect a \\textbf{H}ypertext \\textbf{E}ntity \\textbf{E}xtraction \\textbf{D}ataset (\\textit{HEED}) from the e-commerce domains, scraping both the text and t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.01698","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-04T03:21:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e577ddf7f4700c4fdc7bab061b6c9bb95c963810cfd9c9c549de0ee42d171ce2","abstract_canon_sha256":"01ad2047bbd705ede06878ba708f7fea8803fb66aedcd43361eb73faeb77e8bb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:50.563223Z","signature_b64":"uOXblrMmHooRGOegwUfecDD7dcYgwBFus9UGGvVwkz67DXo5LI44E34pLS6Uc/SoaKnuMRZQBejGyXXkG1vODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"814e9eab318ac8c48c7d36dd0b1015586020abb3e651376582ed1d29b541d621","last_reissued_at":"2026-07-05T07:51:50.562826Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:50.562826Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hypertext Entity Extraction in Webpage","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bo Shao, Daxin Jiang, Hai Zhao, Linjun Shou, Ming Gong, Tianqiao Liu, Yifei Yang","submitted_at":"2024-03-04T03:21:40Z","abstract_excerpt":"Webpage entity extraction is a fundamental natural language processing task in both research and applications. Nowadays, the majority of webpage entity extraction models are trained on structured datasets which strive to retain textual content and its structure information. However, existing datasets all overlook the rich hypertext features (e.g., font color, font size) which show their effectiveness in previous works. To this end, we first collect a \\textbf{H}ypertext \\textbf{E}ntity \\textbf{E}xtraction \\textbf{D}ataset (\\textit{HEED}) from the e-commerce domains, scraping both the text and t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.01698","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.01698/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.01698","created_at":"2026-07-05T07:51:50.562880+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.01698v1","created_at":"2026-07-05T07:51:50.562880+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.01698","created_at":"2026-07-05T07:51:50.562880+00:00"},{"alias_kind":"pith_short_12","alias_value":"QFHJ5KZRRLEM","created_at":"2026-07-05T07:51:50.562880+00:00"},{"alias_kind":"pith_short_16","alias_value":"QFHJ5KZRRLEMJDD5","created_at":"2026-07-05T07:51:50.562880+00:00"},{"alias_kind":"pith_short_8","alias_value":"QFHJ5KZR","created_at":"2026-07-05T07:51:50.562880+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB","json":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB.json","graph_json":"https://pith.science/api/pith-number/QFHJ5KZRRLEMJDD5G3OQWEAVLB/graph.json","events_json":"https://pith.science/api/pith-number/QFHJ5KZRRLEMJDD5G3OQWEAVLB/events.json","paper":"https://pith.science/paper/QFHJ5KZR"},"agent_actions":{"view_html":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB","download_json":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB.json","view_paper":"https://pith.science/paper/QFHJ5KZR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.01698&json=true","fetch_graph":"https://pith.science/api/pith-number/QFHJ5KZRRLEMJDD5G3OQWEAVLB/graph.json","fetch_events":"https://pith.science/api/pith-number/QFHJ5KZRRLEMJDD5G3OQWEAVLB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB/action/storage_attestation","attest_author":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB/action/author_attestation","sign_citation":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB/action/citation_signature","submit_replication":"https://pith.science/pith/QFHJ5KZRRLEMJDD5G3OQWEAVLB/action/replication_record"}},"created_at":"2026-07-05T07:51:50.562880+00:00","updated_at":"2026-07-05T07:51:50.562880+00:00"}