{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PT6PKMY7GWB4BR2FLCXBAK5KFZ","short_pith_number":"pith:PT6PKMY7","schema_version":"1.0","canonical_sha256":"7cfcf5331f3583c0c74558ae102baa2e6a57be9e3e279ec8996c92ad10669ab5","source":{"kind":"arxiv","id":"2410.09992","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Gender Bias of LLMs in Making Morality Judgements","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Divij Bajaj, Jonathan Tong, Ruihong Huang, Yuanyuan Lei","submitted_at":"2024-10-13T20:19:11Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in a multitude of Natural Language Processing (NLP) tasks. However, these models are still not immune to limitations such as social biases, especially gender bias. This work investigates whether current closed and open-source LLMs possess gender bias, especially when asked to give moral opinions. To evaluate these models, we curate and introduce a new dataset GenMO (Gender-bias in Morality Opinions) comprising parallel short stories featuring male and female characters respectively. Specifically, we test models from the GPT family"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09992","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-13T20:19:11Z","cross_cats_sorted":[],"title_canon_sha256":"edb97525fde3f59dcab6bb425f803717123d8eb8bf0a2590ad8ec48862dc26a7","abstract_canon_sha256":"01a6a1768019318b13a57c8a5be498df5f02f15eb6c690ddbcc356572d3e5931"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:09.115357Z","signature_b64":"WoajArSTY42k3M/P7fxxu1b2/l3vjJlHW0ojEunXGrwGB/XO2RFuOD9zXMlLmNdsASHpP9eH0RPFxa47dFoyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7cfcf5331f3583c0c74558ae102baa2e6a57be9e3e279ec8996c92ad10669ab5","last_reissued_at":"2026-07-05T09:20:09.114876Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:09.114876Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Gender Bias of LLMs in Making Morality Judgements","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Divij Bajaj, Jonathan Tong, Ruihong Huang, Yuanyuan Lei","submitted_at":"2024-10-13T20:19:11Z","abstract_excerpt":"Large Language Models (LLMs) have shown remarkable capabilities in a multitude of Natural Language Processing (NLP) tasks. However, these models are still not immune to limitations such as social biases, especially gender bias. This work investigates whether current closed and open-source LLMs possess gender bias, especially when asked to give moral opinions. To evaluate these models, we curate and introduce a new dataset GenMO (Gender-bias in Morality Opinions) comprising parallel short stories featuring male and female characters respectively. Specifically, we test models from the GPT family"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09992","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09992/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09992","created_at":"2026-07-05T09:20:09.114936+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09992v1","created_at":"2026-07-05T09:20:09.114936+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09992","created_at":"2026-07-05T09:20:09.114936+00:00"},{"alias_kind":"pith_short_12","alias_value":"PT6PKMY7GWB4","created_at":"2026-07-05T09:20:09.114936+00:00"},{"alias_kind":"pith_short_16","alias_value":"PT6PKMY7GWB4BR2F","created_at":"2026-07-05T09:20:09.114936+00:00"},{"alias_kind":"pith_short_8","alias_value":"PT6PKMY7","created_at":"2026-07-05T09:20:09.114936+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.03627","citing_title":"Unequal Verdicts: Investigating Gender Bias in LLM-Based Fake News Detection","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ","json":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ.json","graph_json":"https://pith.science/api/pith-number/PT6PKMY7GWB4BR2FLCXBAK5KFZ/graph.json","events_json":"https://pith.science/api/pith-number/PT6PKMY7GWB4BR2FLCXBAK5KFZ/events.json","paper":"https://pith.science/paper/PT6PKMY7"},"agent_actions":{"view_html":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ","download_json":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ.json","view_paper":"https://pith.science/paper/PT6PKMY7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09992&json=true","fetch_graph":"https://pith.science/api/pith-number/PT6PKMY7GWB4BR2FLCXBAK5KFZ/graph.json","fetch_events":"https://pith.science/api/pith-number/PT6PKMY7GWB4BR2FLCXBAK5KFZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ/action/storage_attestation","attest_author":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ/action/author_attestation","sign_citation":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ/action/citation_signature","submit_replication":"https://pith.science/pith/PT6PKMY7GWB4BR2FLCXBAK5KFZ/action/replication_record"}},"created_at":"2026-07-05T09:20:09.114936+00:00","updated_at":"2026-07-05T09:20:09.114936+00:00"}