{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:U3NRK4GLKZV5TFMZ6AXYWK7CL2","short_pith_number":"pith:U3NRK4GL","schema_version":"1.0","canonical_sha256":"a6db1570cb566bd99599f02f8b2be25eaa9c8bc98b7b123b7696fc0fb5d233a8","source":{"kind":"arxiv","id":"2212.09747","version":2},"attestation_state":"computed","paper":{"title":"Do CoNLL-2003 Named Entity Taggers Still Work Well in 2023?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alan Ritter, Shuheng Liu","submitted_at":"2022-12-19T18:59:56Z","abstract_excerpt":"The CoNLL-2003 English named entity recognition (NER) dataset has been widely used to train and evaluate NER models for almost 20 years. However, it is unclear how well models that are trained on this 20-year-old data and developed over a period of decades using the same test set will perform when applied on modern data. In this paper, we evaluate the generalization of over 20 different models trained on CoNLL-2003, and show that NER models have very different generalization. Surprisingly, we find no evidence of performance degradation in pre-trained Transformers, such as RoBERTa and T5, even "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.09747","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-12-19T18:59:56Z","cross_cats_sorted":[],"title_canon_sha256":"13c7ea4ba422317bc86623b5e123cb7a8ba1dd63474fcb2ca7688d1de06ac929","abstract_canon_sha256":"d1bddfa849e383843a2410ce609dd0b4bc9a62bfe6e8a031de3f2bc91d138ee6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:30:03.420100Z","signature_b64":"7wo2l1MuXqHL4TX9mjv75nJV45dQ6qzBe7u+ojCl/1x8pUacN28+H49zAfCsqkcOCDerXEfM9284IiI+pK3zBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6db1570cb566bd99599f02f8b2be25eaa9c8bc98b7b123b7696fc0fb5d233a8","last_reissued_at":"2026-07-05T06:30:03.419746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:30:03.419746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do CoNLL-2003 Named Entity Taggers Still Work Well in 2023?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alan Ritter, Shuheng Liu","submitted_at":"2022-12-19T18:59:56Z","abstract_excerpt":"The CoNLL-2003 English named entity recognition (NER) dataset has been widely used to train and evaluate NER models for almost 20 years. However, it is unclear how well models that are trained on this 20-year-old data and developed over a period of decades using the same test set will perform when applied on modern data. In this paper, we evaluate the generalization of over 20 different models trained on CoNLL-2003, and show that NER models have very different generalization. Surprisingly, we find no evidence of performance degradation in pre-trained Transformers, such as RoBERTa and T5, even "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.09747","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.09747/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.09747","created_at":"2026-07-05T06:30:03.419802+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.09747v2","created_at":"2026-07-05T06:30:03.419802+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.09747","created_at":"2026-07-05T06:30:03.419802+00:00"},{"alias_kind":"pith_short_12","alias_value":"U3NRK4GLKZV5","created_at":"2026-07-05T06:30:03.419802+00:00"},{"alias_kind":"pith_short_16","alias_value":"U3NRK4GLKZV5TFMZ","created_at":"2026-07-05T06:30:03.419802+00:00"},{"alias_kind":"pith_short_8","alias_value":"U3NRK4GL","created_at":"2026-07-05T06:30:03.419802+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.01754","citing_title":"Unraveling Media Perspectives: A Comprehensive Methodology Combining Large Language Models, Topic Modeling, Sentiment Analysis, and Ontology Learning to Analyse Media Bias","ref_index":65,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2","json":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2.json","graph_json":"https://pith.science/api/pith-number/U3NRK4GLKZV5TFMZ6AXYWK7CL2/graph.json","events_json":"https://pith.science/api/pith-number/U3NRK4GLKZV5TFMZ6AXYWK7CL2/events.json","paper":"https://pith.science/paper/U3NRK4GL"},"agent_actions":{"view_html":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2","download_json":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2.json","view_paper":"https://pith.science/paper/U3NRK4GL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.09747&json=true","fetch_graph":"https://pith.science/api/pith-number/U3NRK4GLKZV5TFMZ6AXYWK7CL2/graph.json","fetch_events":"https://pith.science/api/pith-number/U3NRK4GLKZV5TFMZ6AXYWK7CL2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2/action/storage_attestation","attest_author":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2/action/author_attestation","sign_citation":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2/action/citation_signature","submit_replication":"https://pith.science/pith/U3NRK4GLKZV5TFMZ6AXYWK7CL2/action/replication_record"}},"created_at":"2026-07-05T06:30:03.419802+00:00","updated_at":"2026-07-05T06:30:03.419802+00:00"}