{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:GTYXIXFEXFY45EN4VWQKACPZJP","short_pith_number":"pith:GTYXIXFE","schema_version":"1.0","canonical_sha256":"34f1745ca4b971ce91bcada0a009f94bdb66a97a70018b296b40b300a486416b","source":{"kind":"arxiv","id":"2104.08646","version":3},"attestation_state":"computed","paper":{"title":"Competency Problems: On Finding and Removing Artifacts in Language Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexis Ross, Jesse Dodge, Matt Gardner, Matthew E. Peters, Noah A. Smith, Sameer Singh, William Merrill","submitted_at":"2021-04-17T21:34:10Z","abstract_excerpt":"Much recent work in NLP has documented dataset artifacts, bias, and spurious correlations between input features and output labels. However, how to tell which features have \"spurious\" instead of legitimate correlations is typically left unspecified. In this work we argue that for complex language understanding tasks, all simple feature correlations are spurious, and we formalize this notion into a class of problems which we call competency problems. For example, the word \"amazing\" on its own should not give information about a sentiment label independent of the context in which it appears, whi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.08646","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-17T21:34:10Z","cross_cats_sorted":[],"title_canon_sha256":"a0eb7b5ac7a89e9c47d6b3c04c5231cbdd8cdfd8c5112bc11e24dc2f5ca6483c","abstract_canon_sha256":"9bc162731be871e2cc34b06f5277cfa2725e78a999215cef339ea570868263ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:44:09.834411Z","signature_b64":"eYglD+Zw/Mz7WNxDoP2yCMd1TFy2w1rZfrOPz+j6BNoPrw3FOMfCOdH2A3odlRF4/iRmAASZAnrrFayI5iITDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34f1745ca4b971ce91bcada0a009f94bdb66a97a70018b296b40b300a486416b","last_reissued_at":"2026-07-05T03:44:09.834003Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:44:09.834003Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Competency Problems: On Finding and Removing Artifacts in Language Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexis Ross, Jesse Dodge, Matt Gardner, Matthew E. Peters, Noah A. Smith, Sameer Singh, William Merrill","submitted_at":"2021-04-17T21:34:10Z","abstract_excerpt":"Much recent work in NLP has documented dataset artifacts, bias, and spurious correlations between input features and output labels. However, how to tell which features have \"spurious\" instead of legitimate correlations is typically left unspecified. In this work we argue that for complex language understanding tasks, all simple feature correlations are spurious, and we formalize this notion into a class of problems which we call competency problems. For example, the word \"amazing\" on its own should not give information about a sentiment label independent of the context in which it appears, whi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.08646","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.08646/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.08646","created_at":"2026-07-05T03:44:09.834052+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.08646v3","created_at":"2026-07-05T03:44:09.834052+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.08646","created_at":"2026-07-05T03:44:09.834052+00:00"},{"alias_kind":"pith_short_12","alias_value":"GTYXIXFEXFY4","created_at":"2026-07-05T03:44:09.834052+00:00"},{"alias_kind":"pith_short_16","alias_value":"GTYXIXFEXFY45EN4","created_at":"2026-07-05T03:44:09.834052+00:00"},{"alias_kind":"pith_short_8","alias_value":"GTYXIXFE","created_at":"2026-07-05T03:44:09.834052+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.21302","citing_title":"Can human clinical rationales improve the performance and explainability of clinical text classification models?","ref_index":37,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP","json":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP.json","graph_json":"https://pith.science/api/pith-number/GTYXIXFEXFY45EN4VWQKACPZJP/graph.json","events_json":"https://pith.science/api/pith-number/GTYXIXFEXFY45EN4VWQKACPZJP/events.json","paper":"https://pith.science/paper/GTYXIXFE"},"agent_actions":{"view_html":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP","download_json":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP.json","view_paper":"https://pith.science/paper/GTYXIXFE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.08646&json=true","fetch_graph":"https://pith.science/api/pith-number/GTYXIXFEXFY45EN4VWQKACPZJP/graph.json","fetch_events":"https://pith.science/api/pith-number/GTYXIXFEXFY45EN4VWQKACPZJP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP/action/storage_attestation","attest_author":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP/action/author_attestation","sign_citation":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP/action/citation_signature","submit_replication":"https://pith.science/pith/GTYXIXFEXFY45EN4VWQKACPZJP/action/replication_record"}},"created_at":"2026-07-05T03:44:09.834052+00:00","updated_at":"2026-07-05T03:44:09.834052+00:00"}