{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VYKXJJCSNJAK6V66HDYR3GXYMY","short_pith_number":"pith:VYKXJJCS","schema_version":"1.0","canonical_sha256":"ae1574a4526a40af57de38f11d9af8661f066271b682211eeceaf06344bcde9f","source":{"kind":"arxiv","id":"2406.08183","version":2},"attestation_state":"computed","paper":{"title":"Underneath the Numbers: Quantitative and Qualitative Gender Fairness in LLMs for Depression Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hatice Gunes, Jiaee Cheong, Micol Spitale","submitted_at":"2024-06-12T13:14:19Z","abstract_excerpt":"Recent studies show bias in many machine learning models for depression detection, but bias in LLMs for this task remains unexplored. This work presents the first attempt to investigate the degree of gender bias present in existing LLMs (ChatGPT, LLaMA 2, and Bard) using both quantitative and qualitative approaches. From our quantitative evaluation, we found that ChatGPT performs the best across various performance metrics and LLaMA 2 outperforms other LLMs in terms of group fairness metrics. As qualitative fairness evaluation remains an open research question we propose several strategies (e."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.08183","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-12T13:14:19Z","cross_cats_sorted":[],"title_canon_sha256":"90300e2b203fc15966429ee56a6d64fd9f3e045c7230dba4d6ff322e46011e6e","abstract_canon_sha256":"98f1525c7e6bf1433748384b25dac07b8a7e2efa6648b28d3126ab4b2cb0dec1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:53.024351Z","signature_b64":"TzJRCqnBx2fgrB8l/XMagnCdWHCbvUBlFoDzzxAXRoQnBMgI7SnBJYfLfcDUaWI6VjDKQfbzEHS9rG3e7+bzAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae1574a4526a40af57de38f11d9af8661f066271b682211eeceaf06344bcde9f","last_reissued_at":"2026-07-05T08:31:53.023829Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:53.023829Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Underneath the Numbers: Quantitative and Qualitative Gender Fairness in LLMs for Depression Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hatice Gunes, Jiaee Cheong, Micol Spitale","submitted_at":"2024-06-12T13:14:19Z","abstract_excerpt":"Recent studies show bias in many machine learning models for depression detection, but bias in LLMs for this task remains unexplored. This work presents the first attempt to investigate the degree of gender bias present in existing LLMs (ChatGPT, LLaMA 2, and Bard) using both quantitative and qualitative approaches. From our quantitative evaluation, we found that ChatGPT performs the best across various performance metrics and LLaMA 2 outperforms other LLMs in terms of group fairness metrics. As qualitative fairness evaluation remains an open research question we propose several strategies (e."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.08183","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.08183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.08183","created_at":"2026-07-05T08:31:53.023893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.08183v2","created_at":"2026-07-05T08:31:53.023893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.08183","created_at":"2026-07-05T08:31:53.023893+00:00"},{"alias_kind":"pith_short_12","alias_value":"VYKXJJCSNJAK","created_at":"2026-07-05T08:31:53.023893+00:00"},{"alias_kind":"pith_short_16","alias_value":"VYKXJJCSNJAK6V66","created_at":"2026-07-05T08:31:53.023893+00:00"},{"alias_kind":"pith_short_8","alias_value":"VYKXJJCS","created_at":"2026-07-05T08:31:53.023893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23786","citing_title":"FAIR_XAI: Improving Multimodal Foundation Model Fairness via Explainability for Wellbeing Assessment","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY","json":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY.json","graph_json":"https://pith.science/api/pith-number/VYKXJJCSNJAK6V66HDYR3GXYMY/graph.json","events_json":"https://pith.science/api/pith-number/VYKXJJCSNJAK6V66HDYR3GXYMY/events.json","paper":"https://pith.science/paper/VYKXJJCS"},"agent_actions":{"view_html":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY","download_json":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY.json","view_paper":"https://pith.science/paper/VYKXJJCS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.08183&json=true","fetch_graph":"https://pith.science/api/pith-number/VYKXJJCSNJAK6V66HDYR3GXYMY/graph.json","fetch_events":"https://pith.science/api/pith-number/VYKXJJCSNJAK6V66HDYR3GXYMY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY/action/storage_attestation","attest_author":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY/action/author_attestation","sign_citation":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY/action/citation_signature","submit_replication":"https://pith.science/pith/VYKXJJCSNJAK6V66HDYR3GXYMY/action/replication_record"}},"created_at":"2026-07-05T08:31:53.023893+00:00","updated_at":"2026-07-05T08:31:53.023893+00:00"}