{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NJX2EYO4ZHQS3U7Z2WGWW5C2AW","short_pith_number":"pith:NJX2EYO4","schema_version":"1.0","canonical_sha256":"6a6fa261dcc9e12dd3f9d58d6b745a05a92468962a6e689d2035df4583529abc","source":{"kind":"arxiv","id":"2309.09697","version":3},"attestation_state":"computed","paper":{"title":"Evaluating Gender Bias of Pre-trained Language Models in Natural Language Inference by Considering All Labels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masahiro Kaneko, Naoaki Okazaki, Panatchakorn Anantaprayoon","submitted_at":"2023-09-18T12:02:21Z","abstract_excerpt":"Discriminatory gender biases have been found in Pre-trained Language Models (PLMs) for multiple languages. In Natural Language Inference (NLI), existing bias evaluation methods have focused on the prediction results of one specific label out of three labels, such as neutral. However, such evaluation methods can be inaccurate since unique biased inferences are associated with unique prediction labels. Addressing this limitation, we propose a bias evaluation method for PLMs, called NLI-CoAL, which considers all the three labels of NLI task. First, we create three evaluation data groups that repr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.09697","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-18T12:02:21Z","cross_cats_sorted":[],"title_canon_sha256":"6ab7955cf835bfe4c9d9bde140237c24c782c4712076f896ddfd2d00e3b01739","abstract_canon_sha256":"834adb399605f1e782f4dfd3b595b54c05012b942ffad374fb051dce1b87e093"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:24.447694Z","signature_b64":"xs9zkSGDRtS8VDrhJi2OIWcdxnsI6JyYG/VaX2X/uBGg/YllXZSKkrEXt1+B9wSc7Y2SKbE6JVQdTrI92DT7AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a6fa261dcc9e12dd3f9d58d6b745a05a92468962a6e689d2035df4583529abc","last_reissued_at":"2026-07-05T08:20:24.447276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:24.447276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Gender Bias of Pre-trained Language Models in Natural Language Inference by Considering All Labels","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masahiro Kaneko, Naoaki Okazaki, Panatchakorn Anantaprayoon","submitted_at":"2023-09-18T12:02:21Z","abstract_excerpt":"Discriminatory gender biases have been found in Pre-trained Language Models (PLMs) for multiple languages. In Natural Language Inference (NLI), existing bias evaluation methods have focused on the prediction results of one specific label out of three labels, such as neutral. However, such evaluation methods can be inaccurate since unique biased inferences are associated with unique prediction labels. Addressing this limitation, we propose a bias evaluation method for PLMs, called NLI-CoAL, which considers all the three labels of NLI task. First, we create three evaluation data groups that repr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.09697","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.09697/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.09697","created_at":"2026-07-05T08:20:24.447338+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.09697v3","created_at":"2026-07-05T08:20:24.447338+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.09697","created_at":"2026-07-05T08:20:24.447338+00:00"},{"alias_kind":"pith_short_12","alias_value":"NJX2EYO4ZHQS","created_at":"2026-07-05T08:20:24.447338+00:00"},{"alias_kind":"pith_short_16","alias_value":"NJX2EYO4ZHQS3U7Z","created_at":"2026-07-05T08:20:24.447338+00:00"},{"alias_kind":"pith_short_8","alias_value":"NJX2EYO4","created_at":"2026-07-05T08:20:24.447338+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.07090","citing_title":"BharatBBQ: A Multilingual Bias Benchmark for Question Answering in the Indian Context","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW","json":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW.json","graph_json":"https://pith.science/api/pith-number/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/graph.json","events_json":"https://pith.science/api/pith-number/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/events.json","paper":"https://pith.science/paper/NJX2EYO4"},"agent_actions":{"view_html":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW","download_json":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW.json","view_paper":"https://pith.science/paper/NJX2EYO4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.09697&json=true","fetch_graph":"https://pith.science/api/pith-number/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/graph.json","fetch_events":"https://pith.science/api/pith-number/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/action/storage_attestation","attest_author":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/action/author_attestation","sign_citation":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/action/citation_signature","submit_replication":"https://pith.science/pith/NJX2EYO4ZHQS3U7Z2WGWW5C2AW/action/replication_record"}},"created_at":"2026-07-05T08:20:24.447338+00:00","updated_at":"2026-07-05T08:20:24.447338+00:00"}