{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JONZN4IMLD4VE7S7YKFYPPVKVR","short_pith_number":"pith:JONZN4IM","schema_version":"1.0","canonical_sha256":"4b9b96f10c58f9527e5fc28b87beaaac40a10b552d53003b5872e84df715864c","source":{"kind":"arxiv","id":"2210.04337","version":1},"attestation_state":"computed","paper":{"title":"Quantifying Social Biases Using Templates is Unreliable","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Pouya Pezeshkpour, Preethi Seshadri, Sameer Singh","submitted_at":"2022-10-09T20:05:29Z","abstract_excerpt":"Recently, there has been an increase in efforts to understand how large language models (LLMs) propagate and amplify social biases. Several works have utilized templates for fairness evaluation, which allow researchers to quantify social biases in the absence of test sets with protected attribute labels. While template evaluation can be a convenient and helpful diagnostic tool to understand model deficiencies, it often uses a simplistic and limited set of templates. In this paper, we study whether bias measurements are sensitive to the choice of templates used for benchmarking. Specifically, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.04337","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-09T20:05:29Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e24d56d3d5fa7217ecc07099bb346d2e75d8b80c54323a80c3aad868afd32379","abstract_canon_sha256":"0fc86a43ccc23cc83bd9b1f2fa0da35fe1b8eff5378988f40f207061352a9138"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:52.257965Z","signature_b64":"NdvHip7ZPaMTMJgNvDlhp9khiGVCUhYsTvdbwPrNhtt8gd36FKQRhcmYB0ztOoSjJo2vjyn+eQdMjK1mfRPhCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4b9b96f10c58f9527e5fc28b87beaaac40a10b552d53003b5872e84df715864c","last_reissued_at":"2026-07-05T05:04:52.257542Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:52.257542Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying Social Biases Using Templates is Unreliable","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Pouya Pezeshkpour, Preethi Seshadri, Sameer Singh","submitted_at":"2022-10-09T20:05:29Z","abstract_excerpt":"Recently, there has been an increase in efforts to understand how large language models (LLMs) propagate and amplify social biases. Several works have utilized templates for fairness evaluation, which allow researchers to quantify social biases in the absence of test sets with protected attribute labels. While template evaluation can be a convenient and helpful diagnostic tool to understand model deficiencies, it often uses a simplistic and limited set of templates. In this paper, we study whether bias measurements are sensitive to the choice of templates used for benchmarking. Specifically, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.04337","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.04337/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.04337","created_at":"2026-07-05T05:04:52.257601+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.04337v1","created_at":"2026-07-05T05:04:52.257601+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.04337","created_at":"2026-07-05T05:04:52.257601+00:00"},{"alias_kind":"pith_short_12","alias_value":"JONZN4IMLD4V","created_at":"2026-07-05T05:04:52.257601+00:00"},{"alias_kind":"pith_short_16","alias_value":"JONZN4IMLD4VE7S7","created_at":"2026-07-05T05:04:52.257601+00:00"},{"alias_kind":"pith_short_8","alias_value":"JONZN4IM","created_at":"2026-07-05T05:04:52.257601+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23377","citing_title":"Perspective Dial: Measuring Perspective of Text and Guiding LLM Outputs","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR","json":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR.json","graph_json":"https://pith.science/api/pith-number/JONZN4IMLD4VE7S7YKFYPPVKVR/graph.json","events_json":"https://pith.science/api/pith-number/JONZN4IMLD4VE7S7YKFYPPVKVR/events.json","paper":"https://pith.science/paper/JONZN4IM"},"agent_actions":{"view_html":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR","download_json":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR.json","view_paper":"https://pith.science/paper/JONZN4IM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.04337&json=true","fetch_graph":"https://pith.science/api/pith-number/JONZN4IMLD4VE7S7YKFYPPVKVR/graph.json","fetch_events":"https://pith.science/api/pith-number/JONZN4IMLD4VE7S7YKFYPPVKVR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR/action/storage_attestation","attest_author":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR/action/author_attestation","sign_citation":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR/action/citation_signature","submit_replication":"https://pith.science/pith/JONZN4IMLD4VE7S7YKFYPPVKVR/action/replication_record"}},"created_at":"2026-07-05T05:04:52.257601+00:00","updated_at":"2026-07-05T05:04:52.257601+00:00"}