{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:MXJKX5JT6WM5L5CB36PHBZOKBP","short_pith_number":"pith:MXJKX5JT","schema_version":"1.0","canonical_sha256":"65d2abf533f599d5f441df9e70e5ca0bcb3f9e073c9f9c55f194f37e652ca6d0","source":{"kind":"arxiv","id":"2005.14050","version":2},"attestation_state":"computed","paper":{"title":"Language (Technology) is Power: A Critical Survey of \"Bias\" in NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Hal Daum\\'e III, Hanna Wallach, Solon Barocas, Su Lin Blodgett","submitted_at":"2020-05-28T14:32:08Z","abstract_excerpt":"We survey 146 papers analyzing \"bias\" in NLP systems, finding that their motivations are often vague, inconsistent, and lacking in normative reasoning, despite the fact that analyzing \"bias\" is an inherently normative process. We further find that these papers' proposed quantitative techniques for measuring or mitigating \"bias\" are poorly matched to their motivations and do not engage with the relevant literature outside of NLP. Based on these findings, we describe the beginnings of a path forward by proposing three recommendations that should guide work analyzing \"bias\" in NLP systems. These "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.14050","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-05-28T14:32:08Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"7830079aef9bf496ae15e6e4237e9b1d9b22e12d530a361c7b6f95bbe47a47a0","abstract_canon_sha256":"66635827c1a9d20594a535424e92e23b0bf58b386c1d89f2ecf3bb2f016ec12c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:06:35.139503Z","signature_b64":"Ww7CMS20OmXoy2JXnMtgr2/ZzwdWINzeWAp6cggOSVIDhE6gJJMy30mi0l9bg1ERAn92dGuuZ9xC5I5Uj+tcAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65d2abf533f599d5f441df9e70e5ca0bcb3f9e073c9f9c55f194f37e652ca6d0","last_reissued_at":"2026-07-05T01:06:35.139061Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:06:35.139061Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language (Technology) is Power: A Critical Survey of \"Bias\" in NLP","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Hal Daum\\'e III, Hanna Wallach, Solon Barocas, Su Lin Blodgett","submitted_at":"2020-05-28T14:32:08Z","abstract_excerpt":"We survey 146 papers analyzing \"bias\" in NLP systems, finding that their motivations are often vague, inconsistent, and lacking in normative reasoning, despite the fact that analyzing \"bias\" is an inherently normative process. We further find that these papers' proposed quantitative techniques for measuring or mitigating \"bias\" are poorly matched to their motivations and do not engage with the relevant literature outside of NLP. Based on these findings, we describe the beginnings of a path forward by proposing three recommendations that should guide work analyzing \"bias\" in NLP systems. These "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.14050","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.14050/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.14050","created_at":"2026-07-05T01:06:35.139115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.14050v2","created_at":"2026-07-05T01:06:35.139115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.14050","created_at":"2026-07-05T01:06:35.139115+00:00"},{"alias_kind":"pith_short_12","alias_value":"MXJKX5JT6WM5","created_at":"2026-07-05T01:06:35.139115+00:00"},{"alias_kind":"pith_short_16","alias_value":"MXJKX5JT6WM5L5CB","created_at":"2026-07-05T01:06:35.139115+00:00"},{"alias_kind":"pith_short_8","alias_value":"MXJKX5JT","created_at":"2026-07-05T01:06:35.139115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.16723","citing_title":"AgentFairBench: Do LLM Agents Discriminate When They Act?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00019","citing_title":"LLMs in the Real World: Evaluating \"AI\" in Emergency Contexts","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00334","citing_title":"Isolating LLM Lexical Bias: A Curation-Free Triangulated Metric for Preference-Stage Learning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2408.09049","citing_title":"Inertia in Moral and Value Judgments of Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21919","citing_title":"SDGBiasBench: Benchmarking and Mitigating Vision--Language Models' Biases in Sustainable Development Goals","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06816","citing_title":"How do datasets, developers, and models affect biases in a low-resourced language?: The Case of the Bengali Language","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10442","citing_title":"StereoTales: A Multilingual Framework for Open-Ended Stereotype Discovery in LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2211.09085","citing_title":"Galactica: A Large Language Model for Science","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2305.10403","citing_title":"PaLM 2 Technical Report","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10442","citing_title":"StereoTales: A Multilingual Framework for Open-Ended Stereotype Discovery in LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2112.04359","citing_title":"Ethical and social risks of harm from Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18761","citing_title":"From Tokens to Ties: Network and Discourse Analysis of Web3 Ecosystems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05483","citing_title":"Can We Trust a Black-box LLM? LLM Untrustworthy Boundary Detection via Bias-Diffusion and Multi-Agent Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04735","citing_title":"Lighting Up or Dimming Down? Exploring Dark Patterns of LLMs in Co-Creativity","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2101.00027","citing_title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2005.14165","citing_title":"Language Models are Few-Shot Learners","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02640","citing_title":"Trustworthy AI Suffers from Invariance Conflicts and Causality is The Solution","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08774","citing_title":"GPT-4 Technical Report","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP","json":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP.json","graph_json":"https://pith.science/api/pith-number/MXJKX5JT6WM5L5CB36PHBZOKBP/graph.json","events_json":"https://pith.science/api/pith-number/MXJKX5JT6WM5L5CB36PHBZOKBP/events.json","paper":"https://pith.science/paper/MXJKX5JT"},"agent_actions":{"view_html":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP","download_json":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP.json","view_paper":"https://pith.science/paper/MXJKX5JT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.14050&json=true","fetch_graph":"https://pith.science/api/pith-number/MXJKX5JT6WM5L5CB36PHBZOKBP/graph.json","fetch_events":"https://pith.science/api/pith-number/MXJKX5JT6WM5L5CB36PHBZOKBP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP/action/storage_attestation","attest_author":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP/action/author_attestation","sign_citation":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP/action/citation_signature","submit_replication":"https://pith.science/pith/MXJKX5JT6WM5L5CB36PHBZOKBP/action/replication_record"}},"created_at":"2026-07-05T01:06:35.139115+00:00","updated_at":"2026-07-05T01:06:35.139115+00:00"}