{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DGCU4F4LOUXIU3IGAHMCJKM4SS","short_pith_number":"pith:DGCU4F4L","schema_version":"1.0","canonical_sha256":"19854e178b752e8a6d0601d824a99c948faf2988e700e9022fbf4213e1913682","source":{"kind":"arxiv","id":"2407.03129","version":1},"attestation_state":"computed","paper":{"title":"Social Bias Evaluation for Large Language Models Requires Prompt Variations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masahiro Kaneko, Naoaki Okazaki, Rem Hida","submitted_at":"2024-07-03T14:12:04Z","abstract_excerpt":"Warning: This paper contains examples of stereotypes and biases. Large Language Models (LLMs) exhibit considerable social biases, and various studies have tried to evaluate and mitigate these biases accurately. Previous studies use downstream tasks as prompts to examine the degree of social biases for evaluation and mitigation. While LLMs' output highly depends on prompts, previous studies evaluating and mitigating bias have often relied on a limited variety of prompts. In this paper, we investigate the sensitivity of LLMs when changing prompt variations (task instruction and prompt, few-shot "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03129","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-03T14:12:04Z","cross_cats_sorted":[],"title_canon_sha256":"1c622ff1411f87e4c7c492ccc2061288b89a33903c3767cccba91415c10850c1","abstract_canon_sha256":"436f761cd9409ff25767d7ae20198190a8dee3f8111cfce0397349b6509d6558"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:46.770235Z","signature_b64":"+ln4s9lvNJLeDxbuorT9TMChYsz0GJo5znEe26caHq8IQpUX1OOY8Qu6+obWa11zZOxuMiDcHHkDYWjrxD7iBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19854e178b752e8a6d0601d824a99c948faf2988e700e9022fbf4213e1913682","last_reissued_at":"2026-07-05T08:39:46.769772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:46.769772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Social Bias Evaluation for Large Language Models Requires Prompt Variations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masahiro Kaneko, Naoaki Okazaki, Rem Hida","submitted_at":"2024-07-03T14:12:04Z","abstract_excerpt":"Warning: This paper contains examples of stereotypes and biases. Large Language Models (LLMs) exhibit considerable social biases, and various studies have tried to evaluate and mitigate these biases accurately. Previous studies use downstream tasks as prompts to examine the degree of social biases for evaluation and mitigation. While LLMs' output highly depends on prompts, previous studies evaluating and mitigating bias have often relied on a limited variety of prompts. In this paper, we investigate the sensitivity of LLMs when changing prompt variations (task instruction and prompt, few-shot "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03129","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03129/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03129","created_at":"2026-07-05T08:39:46.769828+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03129v1","created_at":"2026-07-05T08:39:46.769828+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03129","created_at":"2026-07-05T08:39:46.769828+00:00"},{"alias_kind":"pith_short_12","alias_value":"DGCU4F4LOUXI","created_at":"2026-07-05T08:39:46.769828+00:00"},{"alias_kind":"pith_short_16","alias_value":"DGCU4F4LOUXIU3IG","created_at":"2026-07-05T08:39:46.769828+00:00"},{"alias_kind":"pith_short_8","alias_value":"DGCU4F4L","created_at":"2026-07-05T08:39:46.769828+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05804","citing_title":"Can LLMs Be Constrained to the Past? Improving Knowledge Cutoff through Recall-Based Prompting","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01936","citing_title":"The Thin Line Between Comprehension and Persuasion in LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20677","citing_title":"Intersectional Fairness in Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14548","citing_title":"VoxSafeBench: Not Just What Is Said, but Who, How, and Where","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04127","citing_title":"Position: the Stochastic Parrot in the Coal Mine. Model Collapse is a Threat to Low-Resource Communities","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS","json":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS.json","graph_json":"https://pith.science/api/pith-number/DGCU4F4LOUXIU3IGAHMCJKM4SS/graph.json","events_json":"https://pith.science/api/pith-number/DGCU4F4LOUXIU3IGAHMCJKM4SS/events.json","paper":"https://pith.science/paper/DGCU4F4L"},"agent_actions":{"view_html":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS","download_json":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS.json","view_paper":"https://pith.science/paper/DGCU4F4L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03129&json=true","fetch_graph":"https://pith.science/api/pith-number/DGCU4F4LOUXIU3IGAHMCJKM4SS/graph.json","fetch_events":"https://pith.science/api/pith-number/DGCU4F4LOUXIU3IGAHMCJKM4SS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS/action/storage_attestation","attest_author":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS/action/author_attestation","sign_citation":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS/action/citation_signature","submit_replication":"https://pith.science/pith/DGCU4F4LOUXIU3IGAHMCJKM4SS/action/replication_record"}},"created_at":"2026-07-05T08:39:46.769828+00:00","updated_at":"2026-07-05T08:39:46.769828+00:00"}