{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5OK7FYMAMNXTNE2CMAW522KBY2","short_pith_number":"pith:5OK7FYMA","schema_version":"1.0","canonical_sha256":"eb95f2e180636f369342602ddd6941c6a1bb939ca585a1cbc5593e097a20c941","source":{"kind":"arxiv","id":"2104.07496","version":1},"attestation_state":"computed","paper":{"title":"Unmasking the Mask -- Evaluating Social Biases in Masked Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Danushka Bollegala, Masahiro Kaneko","submitted_at":"2021-04-15T14:40:42Z","abstract_excerpt":"Masked Language Models (MLMs) have shown superior performances in numerous downstream NLP tasks when used as text encoders. Unfortunately, MLMs also demonstrate significantly worrying levels of social biases. We show that the previously proposed evaluation metrics for quantifying the social biases in MLMs are problematic due to following reasons: (1) prediction accuracy of the masked tokens itself tend to be low in some MLMs, which raises questions regarding the reliability of the evaluation metrics that use the (pseudo) likelihood of the predicted tokens, and (2) the correlation between the p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.07496","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-15T14:40:42Z","cross_cats_sorted":[],"title_canon_sha256":"8b6f011c1b8082dd963a489aa3ae7a995ab6071b1a12fabf3f9cc01e7146eb72","abstract_canon_sha256":"9c8edafb5baa9fc89a7db33ec63ffe708e34a9d116e2685ae73b851c741f8ebf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:32:28.569875Z","signature_b64":"TzAHICIQBJNqLRcOYj4Ix6Flc9G7hjztxcghL1o0f/dF4mh6pnUWCw1k2o+RtJX8mBG2DfH0NNeY8y+sMfNiDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb95f2e180636f369342602ddd6941c6a1bb939ca585a1cbc5593e097a20c941","last_reissued_at":"2026-07-05T02:32:28.569476Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:32:28.569476Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unmasking the Mask -- Evaluating Social Biases in Masked Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Danushka Bollegala, Masahiro Kaneko","submitted_at":"2021-04-15T14:40:42Z","abstract_excerpt":"Masked Language Models (MLMs) have shown superior performances in numerous downstream NLP tasks when used as text encoders. Unfortunately, MLMs also demonstrate significantly worrying levels of social biases. We show that the previously proposed evaluation metrics for quantifying the social biases in MLMs are problematic due to following reasons: (1) prediction accuracy of the masked tokens itself tend to be low in some MLMs, which raises questions regarding the reliability of the evaluation metrics that use the (pseudo) likelihood of the predicted tokens, and (2) the correlation between the p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.07496","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.07496/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.07496","created_at":"2026-07-05T02:32:28.569533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.07496v1","created_at":"2026-07-05T02:32:28.569533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.07496","created_at":"2026-07-05T02:32:28.569533+00:00"},{"alias_kind":"pith_short_12","alias_value":"5OK7FYMAMNXT","created_at":"2026-07-05T02:32:28.569533+00:00"},{"alias_kind":"pith_short_16","alias_value":"5OK7FYMAMNXTNE2C","created_at":"2026-07-05T02:32:28.569533+00:00"},{"alias_kind":"pith_short_8","alias_value":"5OK7FYMA","created_at":"2026-07-05T02:32:28.569533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16442","citing_title":"Dutch CrowS-Pairs: Adapting a Challenge Dataset for Measuring Social Biases in Language Models for Dutch","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2","json":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2.json","graph_json":"https://pith.science/api/pith-number/5OK7FYMAMNXTNE2CMAW522KBY2/graph.json","events_json":"https://pith.science/api/pith-number/5OK7FYMAMNXTNE2CMAW522KBY2/events.json","paper":"https://pith.science/paper/5OK7FYMA"},"agent_actions":{"view_html":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2","download_json":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2.json","view_paper":"https://pith.science/paper/5OK7FYMA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.07496&json=true","fetch_graph":"https://pith.science/api/pith-number/5OK7FYMAMNXTNE2CMAW522KBY2/graph.json","fetch_events":"https://pith.science/api/pith-number/5OK7FYMAMNXTNE2CMAW522KBY2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2/action/storage_attestation","attest_author":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2/action/author_attestation","sign_citation":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2/action/citation_signature","submit_replication":"https://pith.science/pith/5OK7FYMAMNXTNE2CMAW522KBY2/action/replication_record"}},"created_at":"2026-07-05T02:32:28.569533+00:00","updated_at":"2026-07-05T02:32:28.569533+00:00"}