{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:X2MWEOVAJTLYZHNQCFJET5JSWF","short_pith_number":"pith:X2MWEOVA","schema_version":"1.0","canonical_sha256":"be99623aa04cd78c9db0115249f532b157a4d811ec599f2a7c8914e1d69d8d8b","source":{"kind":"arxiv","id":"2004.04123","version":2},"attestation_state":"computed","paper":{"title":"Entity-Switched Datasets: An Approach to Auditing the In-Domain Robustness of Named Entity Recognition Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ani Nenkova, Byron C. Wallace, Oshin Agarwal, Yinfei Yang","submitted_at":"2020-04-08T17:11:31Z","abstract_excerpt":"Named entity recognition systems perform well on standard datasets comprising English news. But given the paucity of data, it is difficult to draw conclusions about the robustness of systems with respect to recognizing a diverse set of entities. We propose a method for auditing the in-domain robustness of systems, focusing specifically on differences in performance due to the national origin of entities. We create entity-switched datasets, in which named entities in the original texts are replaced by plausible named entities of the same type but of different national origin. We find that state"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.04123","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-08T17:11:31Z","cross_cats_sorted":[],"title_canon_sha256":"04dc82fad15cf21ae2de1db9e363497e54ad38fded92c17a56d42399f890b0ab","abstract_canon_sha256":"35b3f00ba14157acc39c44201c4b09f631297373bc2c3f761dc39424d0a0ec3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:06:44.209146Z","signature_b64":"YXdpUw1xTLA8PlrSu+jtUqcLmD+xMewPME2umetJcV+zBOCfiMguMJ5Ht3Izz1W7uGg7XywL4JXQyvB7BhaCDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be99623aa04cd78c9db0115249f532b157a4d811ec599f2a7c8914e1d69d8d8b","last_reissued_at":"2026-07-05T02:06:44.208707Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:06:44.208707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Entity-Switched Datasets: An Approach to Auditing the In-Domain Robustness of Named Entity Recognition Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ani Nenkova, Byron C. Wallace, Oshin Agarwal, Yinfei Yang","submitted_at":"2020-04-08T17:11:31Z","abstract_excerpt":"Named entity recognition systems perform well on standard datasets comprising English news. But given the paucity of data, it is difficult to draw conclusions about the robustness of systems with respect to recognizing a diverse set of entities. We propose a method for auditing the in-domain robustness of systems, focusing specifically on differences in performance due to the national origin of entities. We create entity-switched datasets, in which named entities in the original texts are replaced by plausible named entities of the same type but of different national origin. We find that state"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.04123","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.04123/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.04123","created_at":"2026-07-05T02:06:44.208765+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.04123v2","created_at":"2026-07-05T02:06:44.208765+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.04123","created_at":"2026-07-05T02:06:44.208765+00:00"},{"alias_kind":"pith_short_12","alias_value":"X2MWEOVAJTLY","created_at":"2026-07-05T02:06:44.208765+00:00"},{"alias_kind":"pith_short_16","alias_value":"X2MWEOVAJTLYZHNQ","created_at":"2026-07-05T02:06:44.208765+00:00"},{"alias_kind":"pith_short_8","alias_value":"X2MWEOVA","created_at":"2026-07-05T02:06:44.208765+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.00713","citing_title":"Observatorio Lazaro: A self-populating database of anglicism usage in the Spanish press","ref_index":58,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF","json":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF.json","graph_json":"https://pith.science/api/pith-number/X2MWEOVAJTLYZHNQCFJET5JSWF/graph.json","events_json":"https://pith.science/api/pith-number/X2MWEOVAJTLYZHNQCFJET5JSWF/events.json","paper":"https://pith.science/paper/X2MWEOVA"},"agent_actions":{"view_html":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF","download_json":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF.json","view_paper":"https://pith.science/paper/X2MWEOVA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.04123&json=true","fetch_graph":"https://pith.science/api/pith-number/X2MWEOVAJTLYZHNQCFJET5JSWF/graph.json","fetch_events":"https://pith.science/api/pith-number/X2MWEOVAJTLYZHNQCFJET5JSWF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF/action/storage_attestation","attest_author":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF/action/author_attestation","sign_citation":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF/action/citation_signature","submit_replication":"https://pith.science/pith/X2MWEOVAJTLYZHNQCFJET5JSWF/action/replication_record"}},"created_at":"2026-07-05T02:06:44.208765+00:00","updated_at":"2026-07-05T02:06:44.208765+00:00"}