{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:RSBUL3H7RTEC7GEMO2Q3QOJS5A","short_pith_number":"pith:RSBUL3H7","schema_version":"1.0","canonical_sha256":"8c8345ecff8cc82f988c76a1b83932e83851818458ca81eeb4ed8365d25f2fd8","source":{"kind":"arxiv","id":"2112.14168","version":1},"attestation_state":"computed","paper":{"title":"A Survey on Gender Bias in Natural Language Processing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Isabelle Augenstein, Karolina Stanczak","submitted_at":"2021-12-28T14:54:18Z","abstract_excerpt":"Language can be used as a means of reproducing and enforcing harmful stereotypes and biases and has been analysed as such in numerous research. In this paper, we present a survey of 304 papers on gender bias in natural language processing. We analyse definitions of gender and its categories within social sciences and connect them to formal definitions of gender bias in NLP research. We survey lexica and datasets applied in research on gender bias and then compare and contrast approaches to detecting and mitigating gender bias. We find that research on gender bias suffers from four core limitat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.14168","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-12-28T14:54:18Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"e7f07817d52ec1af4ce3b4b03cc18dd262c991287c3642cf7eb753d16d00f5db","abstract_canon_sha256":"4116c3d7fed525eb65dede702582024a6f7aeb1e284aaa99f2e3b68f54c41359"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:44:19.261760Z","signature_b64":"j67Oydz0aRL92NlWuxi8zz6XK/zzkY1YYmZBzqAtiZKPy1fNi3OtaOAnHLANG67izyrhsjWRygZ6UgWUNzByCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c8345ecff8cc82f988c76a1b83932e83851818458ca81eeb4ed8365d25f2fd8","last_reissued_at":"2026-07-05T03:44:19.261281Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:44:19.261281Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Gender Bias in Natural Language Processing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.CL","authors_text":"Isabelle Augenstein, Karolina Stanczak","submitted_at":"2021-12-28T14:54:18Z","abstract_excerpt":"Language can be used as a means of reproducing and enforcing harmful stereotypes and biases and has been analysed as such in numerous research. In this paper, we present a survey of 304 papers on gender bias in natural language processing. We analyse definitions of gender and its categories within social sciences and connect them to formal definitions of gender bias in NLP research. We survey lexica and datasets applied in research on gender bias and then compare and contrast approaches to detecting and mitigating gender bias. We find that research on gender bias suffers from four core limitat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.14168","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.14168/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.14168","created_at":"2026-07-05T03:44:19.261342+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.14168v1","created_at":"2026-07-05T03:44:19.261342+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.14168","created_at":"2026-07-05T03:44:19.261342+00:00"},{"alias_kind":"pith_short_12","alias_value":"RSBUL3H7RTEC","created_at":"2026-07-05T03:44:19.261342+00:00"},{"alias_kind":"pith_short_16","alias_value":"RSBUL3H7RTEC7GEM","created_at":"2026-07-05T03:44:19.261342+00:00"},{"alias_kind":"pith_short_8","alias_value":"RSBUL3H7","created_at":"2026-07-05T03:44:19.261342+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28044","citing_title":"A Tree-of-Thoughts Inspired Hybrid Approach for Legal Case Judgement Summarization using LLMs","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05942","citing_title":"EvalMORAAL: Interpretable Chain-of-Thought and LLM-as-Judge Evaluation for Moral Alignment in Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2512.09292","citing_title":"Identifying Bias in Machine-generated Text Detection","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":186,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":186,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A","json":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A.json","graph_json":"https://pith.science/api/pith-number/RSBUL3H7RTEC7GEMO2Q3QOJS5A/graph.json","events_json":"https://pith.science/api/pith-number/RSBUL3H7RTEC7GEMO2Q3QOJS5A/events.json","paper":"https://pith.science/paper/RSBUL3H7"},"agent_actions":{"view_html":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A","download_json":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A.json","view_paper":"https://pith.science/paper/RSBUL3H7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.14168&json=true","fetch_graph":"https://pith.science/api/pith-number/RSBUL3H7RTEC7GEMO2Q3QOJS5A/graph.json","fetch_events":"https://pith.science/api/pith-number/RSBUL3H7RTEC7GEMO2Q3QOJS5A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A/action/storage_attestation","attest_author":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A/action/author_attestation","sign_citation":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A/action/citation_signature","submit_replication":"https://pith.science/pith/RSBUL3H7RTEC7GEMO2Q3QOJS5A/action/replication_record"}},"created_at":"2026-07-05T03:44:19.261342+00:00","updated_at":"2026-07-05T03:44:19.261342+00:00"}