{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WPSV4UZSTLTT664IC5HTZDMFGN","short_pith_number":"pith:WPSV4UZS","schema_version":"1.0","canonical_sha256":"b3e55e53329ae73f7b88174f3c8d85334ab0fdb97a8a2e78e86833552e26b512","source":{"kind":"arxiv","id":"2210.02938","version":1},"attestation_state":"computed","paper":{"title":"Debiasing isn't enough! -- On the Effectiveness of Debiasing MLMs and their Social Biases in Downstream Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Danushka Bollegala, Masahiro Kaneko, Naoaki Okazaki","submitted_at":"2022-10-06T14:08:57Z","abstract_excerpt":"We study the relationship between task-agnostic intrinsic and task-specific extrinsic social bias evaluation measures for Masked Language Models (MLMs), and find that there exists only a weak correlation between these two types of evaluation measures. Moreover, we find that MLMs debiased using different methods still re-learn social biases during fine-tuning on downstream tasks. We identify the social biases in both training instances as well as their assigned labels as reasons for the discrepancy between intrinsic and extrinsic bias evaluation measurements. Overall, our findings highlight the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.02938","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-06T14:08:57Z","cross_cats_sorted":[],"title_canon_sha256":"fe300f92f22df11217f265b52d0b4658f703e746eb65b37a82ab4747af499966","abstract_canon_sha256":"14d8f2743484480e9641184dba2799679d481cc6fda23c2306f8f72fec551307"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:05.483915Z","signature_b64":"Vo/K+MxVnwQ1sU4vVrryBBU9CapF/4iJF0p/gcJriCt7e0ZzvuzdYCSXgYgC8XSOotGJZTG7s6RUNg8+234pDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b3e55e53329ae73f7b88174f3c8d85334ab0fdb97a8a2e78e86833552e26b512","last_reissued_at":"2026-07-05T05:04:05.483507Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:05.483507Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Debiasing isn't enough! -- On the Effectiveness of Debiasing MLMs and their Social Biases in Downstream Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Danushka Bollegala, Masahiro Kaneko, Naoaki Okazaki","submitted_at":"2022-10-06T14:08:57Z","abstract_excerpt":"We study the relationship between task-agnostic intrinsic and task-specific extrinsic social bias evaluation measures for Masked Language Models (MLMs), and find that there exists only a weak correlation between these two types of evaluation measures. Moreover, we find that MLMs debiased using different methods still re-learn social biases during fine-tuning on downstream tasks. We identify the social biases in both training instances as well as their assigned labels as reasons for the discrepancy between intrinsic and extrinsic bias evaluation measurements. Overall, our findings highlight the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.02938","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.02938/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.02938","created_at":"2026-07-05T05:04:05.483567+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.02938v1","created_at":"2026-07-05T05:04:05.483567+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.02938","created_at":"2026-07-05T05:04:05.483567+00:00"},{"alias_kind":"pith_short_12","alias_value":"WPSV4UZSTLTT","created_at":"2026-07-05T05:04:05.483567+00:00"},{"alias_kind":"pith_short_16","alias_value":"WPSV4UZSTLTT664I","created_at":"2026-07-05T05:04:05.483567+00:00"},{"alias_kind":"pith_short_8","alias_value":"WPSV4UZS","created_at":"2026-07-05T05:04:05.483567+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.01709","citing_title":"Fairness Dynamics During Training","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN","json":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN.json","graph_json":"https://pith.science/api/pith-number/WPSV4UZSTLTT664IC5HTZDMFGN/graph.json","events_json":"https://pith.science/api/pith-number/WPSV4UZSTLTT664IC5HTZDMFGN/events.json","paper":"https://pith.science/paper/WPSV4UZS"},"agent_actions":{"view_html":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN","download_json":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN.json","view_paper":"https://pith.science/paper/WPSV4UZS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.02938&json=true","fetch_graph":"https://pith.science/api/pith-number/WPSV4UZSTLTT664IC5HTZDMFGN/graph.json","fetch_events":"https://pith.science/api/pith-number/WPSV4UZSTLTT664IC5HTZDMFGN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN/action/storage_attestation","attest_author":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN/action/author_attestation","sign_citation":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN/action/citation_signature","submit_replication":"https://pith.science/pith/WPSV4UZSTLTT664IC5HTZDMFGN/action/replication_record"}},"created_at":"2026-07-05T05:04:05.483567+00:00","updated_at":"2026-07-05T05:04:05.483567+00:00"}