{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:EZW34JTV6P3BD77AWQ25DMSBZS","short_pith_number":"pith:EZW34JTV","schema_version":"1.0","canonical_sha256":"266dbe2675f3f611ffe0b435d1b241ccb175a7bfb2e40d42ce2b01e6b13cb323","source":{"kind":"arxiv","id":"1909.01441","version":1},"attestation_state":"computed","paper":{"title":"CrossWeigh: Training Named Entity Tagger from Imperfect Annotations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiacheng Liu, Jiawei Han, Jingbo Shang, Lihao Lu, Liyuan Liu, Zihan Wang","submitted_at":"2019-09-03T20:34:34Z","abstract_excerpt":"Everyone makes mistakes. So do human annotators when curating labels for named entity recognition (NER). Such label mistakes might hurt model training and interfere model comparison. In this study, we dive deep into one of the widely-adopted NER benchmark datasets, CoNLL03 NER. We are able to identify label mistakes in about 5.38% test sentences, which is a significant ratio considering that the state-of-the-art test F1 score is already around 93%. Therefore, we manually correct these label mistakes and form a cleaner test set. Our re-evaluation of popular models on this corrected test set lea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.01441","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-09-03T20:34:34Z","cross_cats_sorted":[],"title_canon_sha256":"17967691fa7379ec27466480002fafa738e89c6bcaa641239b96aaae77a71814","abstract_canon_sha256":"f3becc308598a5c12fa653acaf841842f48f965adb52fede2e0fbbd4c646d784"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:16.382526Z","signature_b64":"FBJPsuQvrjb41/tii2i6JNLKvMlJvyqq1xFJa2UOckYaZ4LF5X1nKQRCkcn93A03jp7t6GvgaKTpr+QMgBeBAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"266dbe2675f3f611ffe0b435d1b241ccb175a7bfb2e40d42ce2b01e6b13cb323","last_reissued_at":"2026-07-05T00:02:16.382028Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:16.382028Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CrossWeigh: Training Named Entity Tagger from Imperfect Annotations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiacheng Liu, Jiawei Han, Jingbo Shang, Lihao Lu, Liyuan Liu, Zihan Wang","submitted_at":"2019-09-03T20:34:34Z","abstract_excerpt":"Everyone makes mistakes. So do human annotators when curating labels for named entity recognition (NER). Such label mistakes might hurt model training and interfere model comparison. In this study, we dive deep into one of the widely-adopted NER benchmark datasets, CoNLL03 NER. We are able to identify label mistakes in about 5.38% test sentences, which is a significant ratio considering that the state-of-the-art test F1 score is already around 93%. Therefore, we manually correct these label mistakes and form a cleaner test set. Our re-evaluation of popular models on this corrected test set lea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.01441","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.01441/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.01441","created_at":"2026-07-05T00:02:16.382103+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.01441v1","created_at":"2026-07-05T00:02:16.382103+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.01441","created_at":"2026-07-05T00:02:16.382103+00:00"},{"alias_kind":"pith_short_12","alias_value":"EZW34JTV6P3B","created_at":"2026-07-05T00:02:16.382103+00:00"},{"alias_kind":"pith_short_16","alias_value":"EZW34JTV6P3BD77A","created_at":"2026-07-05T00:02:16.382103+00:00"},{"alias_kind":"pith_short_8","alias_value":"EZW34JTV","created_at":"2026-07-05T00:02:16.382103+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.11665","citing_title":"Multilingual Prompt Engineering in Large Language Models: A Survey Across NLP Tasks","ref_index":77,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS","json":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS.json","graph_json":"https://pith.science/api/pith-number/EZW34JTV6P3BD77AWQ25DMSBZS/graph.json","events_json":"https://pith.science/api/pith-number/EZW34JTV6P3BD77AWQ25DMSBZS/events.json","paper":"https://pith.science/paper/EZW34JTV"},"agent_actions":{"view_html":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS","download_json":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS.json","view_paper":"https://pith.science/paper/EZW34JTV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.01441&json=true","fetch_graph":"https://pith.science/api/pith-number/EZW34JTV6P3BD77AWQ25DMSBZS/graph.json","fetch_events":"https://pith.science/api/pith-number/EZW34JTV6P3BD77AWQ25DMSBZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS/action/storage_attestation","attest_author":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS/action/author_attestation","sign_citation":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS/action/citation_signature","submit_replication":"https://pith.science/pith/EZW34JTV6P3BD77AWQ25DMSBZS/action/replication_record"}},"created_at":"2026-07-05T00:02:16.382103+00:00","updated_at":"2026-07-05T00:02:16.382103+00:00"}