{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XBOFKHVE5BF5UJVUPYOCV3TU6U","short_pith_number":"pith:XBOFKHVE","schema_version":"1.0","canonical_sha256":"b85c551ea4e84bda26b47e1c2aee74f5346f577878e5ca842ad204e5e79ff463","source":{"kind":"arxiv","id":"2501.14457","version":1},"attestation_state":"computed","paper":{"title":"Understanding and Mitigating Gender Bias in LLMs via Interpretable Neuron Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Sophia Ananiadou, Zeping Yu","submitted_at":"2025-01-24T12:41:30Z","abstract_excerpt":"Large language models (LLMs) often exhibit gender bias, posing challenges for their safe deployment. Existing methods to mitigate bias lack a comprehensive understanding of its mechanisms or compromise the model's core capabilities. To address these issues, we propose the CommonWords dataset, to systematically evaluate gender bias in LLMs. Our analysis reveals pervasive bias across models and identifies specific neuron circuits, including gender neurons and general neurons, responsible for this behavior. Notably, editing even a small number of general neurons can disrupt the model's overall ca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14457","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-24T12:41:30Z","cross_cats_sorted":[],"title_canon_sha256":"4044a5be09024fc642969fd97d1740803975a72821f7c1112d952bc5ebe4b292","abstract_canon_sha256":"f9821b4e76cf4d692bb9fb9dde7580be8710e9afa641439f3ab5ec2383c5ac50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:01.091045Z","signature_b64":"nIvFVjgZLNnshTbOCxRlJesxTGqitrzmjfx6rkecPA4TixwgLNcjl0ez6WrIyl0ixUcXc4r7w9+xorpm9OLVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b85c551ea4e84bda26b47e1c2aee74f5346f577878e5ca842ad204e5e79ff463","last_reissued_at":"2026-07-05T10:05:01.090602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:01.090602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding and Mitigating Gender Bias in LLMs via Interpretable Neuron Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Sophia Ananiadou, Zeping Yu","submitted_at":"2025-01-24T12:41:30Z","abstract_excerpt":"Large language models (LLMs) often exhibit gender bias, posing challenges for their safe deployment. Existing methods to mitigate bias lack a comprehensive understanding of its mechanisms or compromise the model's core capabilities. To address these issues, we propose the CommonWords dataset, to systematically evaluate gender bias in LLMs. Our analysis reveals pervasive bias across models and identifies specific neuron circuits, including gender neurons and general neurons, responsible for this behavior. Notably, editing even a small number of general neurons can disrupt the model's overall ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14457","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14457","created_at":"2026-07-05T10:05:01.090653+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14457v1","created_at":"2026-07-05T10:05:01.090653+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14457","created_at":"2026-07-05T10:05:01.090653+00:00"},{"alias_kind":"pith_short_12","alias_value":"XBOFKHVE5BF5","created_at":"2026-07-05T10:05:01.090653+00:00"},{"alias_kind":"pith_short_16","alias_value":"XBOFKHVE5BF5UJVU","created_at":"2026-07-05T10:05:01.090653+00:00"},{"alias_kind":"pith_short_8","alias_value":"XBOFKHVE","created_at":"2026-07-05T10:05:01.090653+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.03058","citing_title":"Neuron-Anchored Rule Extraction for Large Language Models via Contrastive Hierarchical Ablation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12299","citing_title":"GKnow: Measuring the Entanglement of Gender Bias and Factual Gender","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15294","citing_title":"How Do LLMs and VLMs Understand Viewpoint Rotation Without Vision? An Interpretability Study","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03058","citing_title":"Neuron-Anchored Rule Extraction for Large Language Models via Contrastive Hierarchical Ablation","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U","json":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U.json","graph_json":"https://pith.science/api/pith-number/XBOFKHVE5BF5UJVUPYOCV3TU6U/graph.json","events_json":"https://pith.science/api/pith-number/XBOFKHVE5BF5UJVUPYOCV3TU6U/events.json","paper":"https://pith.science/paper/XBOFKHVE"},"agent_actions":{"view_html":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U","download_json":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U.json","view_paper":"https://pith.science/paper/XBOFKHVE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14457&json=true","fetch_graph":"https://pith.science/api/pith-number/XBOFKHVE5BF5UJVUPYOCV3TU6U/graph.json","fetch_events":"https://pith.science/api/pith-number/XBOFKHVE5BF5UJVUPYOCV3TU6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U/action/storage_attestation","attest_author":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U/action/author_attestation","sign_citation":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U/action/citation_signature","submit_replication":"https://pith.science/pith/XBOFKHVE5BF5UJVUPYOCV3TU6U/action/replication_record"}},"created_at":"2026-07-05T10:05:01.090653+00:00","updated_at":"2026-07-05T10:05:01.090653+00:00"}