{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BBJDWJ36XH73R3PW477466XNDY","short_pith_number":"pith:BBJDWJ36","schema_version":"1.0","canonical_sha256":"08523b277eb9ffb8edf6e7ffcf7aed1e16330f399660fc784db5ec28d8385b6f","source":{"kind":"arxiv","id":"2502.12858","version":1},"attestation_state":"computed","paper":{"title":"Rejected Dialects: Biases Against African American Language in Reward Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Chrysoula Zerva, Daniel Chechelnitsky, Joel Mire, Maarten Sap, Nicholas Deas, Zubin Trivadi Aysola","submitted_at":"2025-02-18T13:45:42Z","abstract_excerpt":"Preference alignment via reward models helps build safe, helpful, and reliable large language models (LLMs). However, subjectivity in preference judgments and the lack of representative sampling in preference data collection can introduce new biases, hindering reward models' fairness and equity. In this work, we introduce a framework for evaluating dialect biases in reward models and conduct a case study on biases against African American Language (AAL) through several experiments comparing reward model preferences and behavior on paired White Mainstream English (WME) and both machine-translat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.12858","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-18T13:45:42Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"2425407964071622280d0725f4dcfb61d42c3dcc65dca7947e67ad13b9397bdb","abstract_canon_sha256":"83eb13a3aed54d2902a02c7c6b0d19716c3c67a80c52065804c2b85015719552"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:25.364497Z","signature_b64":"2fBX/pJeGm4oBEOcfoZxscONx2UB1jSiVtxTkA3npJm1k4CTbLVLLeH/7DIZmftsWqii7MkCWwMkdARonDvACw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"08523b277eb9ffb8edf6e7ffcf7aed1e16330f399660fc784db5ec28d8385b6f","last_reissued_at":"2026-07-05T10:16:25.363965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:25.363965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rejected Dialects: Biases Against African American Language in Reward Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Chrysoula Zerva, Daniel Chechelnitsky, Joel Mire, Maarten Sap, Nicholas Deas, Zubin Trivadi Aysola","submitted_at":"2025-02-18T13:45:42Z","abstract_excerpt":"Preference alignment via reward models helps build safe, helpful, and reliable large language models (LLMs). However, subjectivity in preference judgments and the lack of representative sampling in preference data collection can introduce new biases, hindering reward models' fairness and equity. In this work, we introduce a framework for evaluating dialect biases in reward models and conduct a case study on biases against African American Language (AAL) through several experiments comparing reward model preferences and behavior on paired White Mainstream English (WME) and both machine-translat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.12858","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.12858/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.12858","created_at":"2026-07-05T10:16:25.364022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.12858v1","created_at":"2026-07-05T10:16:25.364022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.12858","created_at":"2026-07-05T10:16:25.364022+00:00"},{"alias_kind":"pith_short_12","alias_value":"BBJDWJ36XH73","created_at":"2026-07-05T10:16:25.364022+00:00"},{"alias_kind":"pith_short_16","alias_value":"BBJDWJ36XH73R3PW","created_at":"2026-07-05T10:16:25.364022+00:00"},{"alias_kind":"pith_short_8","alias_value":"BBJDWJ36","created_at":"2026-07-05T10:16:25.364022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.21774","citing_title":"Probing Latent Colombian Identity Inferences in Qwen2.5-7B with Natural Language Autoencoders","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY","json":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY.json","graph_json":"https://pith.science/api/pith-number/BBJDWJ36XH73R3PW477466XNDY/graph.json","events_json":"https://pith.science/api/pith-number/BBJDWJ36XH73R3PW477466XNDY/events.json","paper":"https://pith.science/paper/BBJDWJ36"},"agent_actions":{"view_html":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY","download_json":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY.json","view_paper":"https://pith.science/paper/BBJDWJ36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.12858&json=true","fetch_graph":"https://pith.science/api/pith-number/BBJDWJ36XH73R3PW477466XNDY/graph.json","fetch_events":"https://pith.science/api/pith-number/BBJDWJ36XH73R3PW477466XNDY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY/action/storage_attestation","attest_author":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY/action/author_attestation","sign_citation":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY/action/citation_signature","submit_replication":"https://pith.science/pith/BBJDWJ36XH73R3PW477466XNDY/action/replication_record"}},"created_at":"2026-07-05T10:16:25.364022+00:00","updated_at":"2026-07-05T10:16:25.364022+00:00"}