{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MH4KACPOZS2XQ36WD2RJ53KE5Q","short_pith_number":"pith:MH4KACPO","schema_version":"1.0","canonical_sha256":"61f8a009eeccb5786fd61ea29eed44ec252dbe188f0914c64d35499f5456f216","source":{"kind":"arxiv","id":"2403.09148","version":1},"attestation_state":"computed","paper":{"title":"Evaluating LLMs for Gender Disparities in Notable Persons","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Arun Sundararajan, Lauren Rhue, Sofie Goethals","submitted_at":"2024-03-14T07:58:27Z","abstract_excerpt":"This study examines the use of Large Language Models (LLMs) for retrieving factual information, addressing concerns over their propensity to produce factually incorrect \"hallucinated\" responses or to altogether decline to even answer prompt at all. Specifically, it investigates the presence of gender-based biases in LLMs' responses to factual inquiries. This paper takes a multi-pronged approach to evaluating GPT models by evaluating fairness across multiple dimensions of recall, hallucinations and declinations. Our findings reveal discernible gender disparities in the responses generated by GP"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.09148","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-14T07:58:27Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"2f8a143c191e0335dc58ba6e759149d17a1ac408b8a29faa44a4a305e7479526","abstract_canon_sha256":"0f9999a42af1e3d5d0cf5c8b1bd08c8cbead87e435093ce4b30d47348f30a707"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:05.965141Z","signature_b64":"vpXjSdE3gqAROd5wTfcqLEFdeW/KsM6jNOnuaokCMKtVcW7WMGqjG4p8SdpeC7mf0evxR+xlGZ8aUdLq8svxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61f8a009eeccb5786fd61ea29eed44ec252dbe188f0914c64d35499f5456f216","last_reissued_at":"2026-07-05T07:56:05.964653Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:05.964653Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating LLMs for Gender Disparities in Notable Persons","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CL","authors_text":"Arun Sundararajan, Lauren Rhue, Sofie Goethals","submitted_at":"2024-03-14T07:58:27Z","abstract_excerpt":"This study examines the use of Large Language Models (LLMs) for retrieving factual information, addressing concerns over their propensity to produce factually incorrect \"hallucinated\" responses or to altogether decline to even answer prompt at all. Specifically, it investigates the presence of gender-based biases in LLMs' responses to factual inquiries. This paper takes a multi-pronged approach to evaluating GPT models by evaluating fairness across multiple dimensions of recall, hallucinations and declinations. Our findings reveal discernible gender disparities in the responses generated by GP"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.09148","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.09148/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.09148","created_at":"2026-07-05T07:56:05.964711+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.09148v1","created_at":"2026-07-05T07:56:05.964711+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.09148","created_at":"2026-07-05T07:56:05.964711+00:00"},{"alias_kind":"pith_short_12","alias_value":"MH4KACPOZS2X","created_at":"2026-07-05T07:56:05.964711+00:00"},{"alias_kind":"pith_short_16","alias_value":"MH4KACPOZS2XQ36W","created_at":"2026-07-05T07:56:05.964711+00:00"},{"alias_kind":"pith_short_8","alias_value":"MH4KACPO","created_at":"2026-07-05T07:56:05.964711+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16989","citing_title":"Obscured but Not Erased: Evaluating Nationality Bias in LLMs via Name-Based Bias Benchmarks","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q","json":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q.json","graph_json":"https://pith.science/api/pith-number/MH4KACPOZS2XQ36WD2RJ53KE5Q/graph.json","events_json":"https://pith.science/api/pith-number/MH4KACPOZS2XQ36WD2RJ53KE5Q/events.json","paper":"https://pith.science/paper/MH4KACPO"},"agent_actions":{"view_html":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q","download_json":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q.json","view_paper":"https://pith.science/paper/MH4KACPO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.09148&json=true","fetch_graph":"https://pith.science/api/pith-number/MH4KACPOZS2XQ36WD2RJ53KE5Q/graph.json","fetch_events":"https://pith.science/api/pith-number/MH4KACPOZS2XQ36WD2RJ53KE5Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q/action/storage_attestation","attest_author":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q/action/author_attestation","sign_citation":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q/action/citation_signature","submit_replication":"https://pith.science/pith/MH4KACPOZS2XQ36WD2RJ53KE5Q/action/replication_record"}},"created_at":"2026-07-05T07:56:05.964711+00:00","updated_at":"2026-07-05T07:56:05.964711+00:00"}