{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:F5HQG4X3AZE3P6G5BI26O5ATHT","short_pith_number":"pith:F5HQG4X3","schema_version":"1.0","canonical_sha256":"2f4f0372fb0649b7f8dd0a35e774133cc1a6790568664ddc2da5a2827d29c802","source":{"kind":"arxiv","id":"1909.01326","version":2},"attestation_state":"computed","paper":{"title":"The Woman Worked as a Babysitter: On Biases in Language Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Emily Sheng, Kai-Wei Chang, Nanyun Peng, Premkumar Natarajan","submitted_at":"2019-09-03T17:50:44Z","abstract_excerpt":"We present a systematic study of biases in natural language generation (NLG) by analyzing text generated from prompts that contain mentions of different demographic groups. In this work, we introduce the notion of the regard towards a demographic, use the varying levels of regard towards different demographics as a defining metric for bias in NLG, and analyze the extent to which sentiment scores are a relevant proxy metric for regard. To this end, we collect strategically-generated text from language models and manually annotate the text with both sentiment and regard scores. Additionally, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.01326","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-09-03T17:50:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"52633b4712c60e68b7fac33960cae68143a316a2699a72b776c5a0a426729f65","abstract_canon_sha256":"8072f27d82f0501a0312bbd2ffb8efc3431cbaf118ef2b313f7313023441ee99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:14:33.565252Z","signature_b64":"ipPYCR1JD4juKkGxw/v4DCYfQjFox7RE94yQlLZt1mVkl8g1yjMlbqf+PayouM4gxI3LMimXZR9vg3nC6nk1DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f4f0372fb0649b7f8dd0a35e774133cc1a6790568664ddc2da5a2827d29c802","last_reissued_at":"2026-07-05T00:14:33.564750Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:14:33.564750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Woman Worked as a Babysitter: On Biases in Language Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Emily Sheng, Kai-Wei Chang, Nanyun Peng, Premkumar Natarajan","submitted_at":"2019-09-03T17:50:44Z","abstract_excerpt":"We present a systematic study of biases in natural language generation (NLG) by analyzing text generated from prompts that contain mentions of different demographic groups. In this work, we introduce the notion of the regard towards a demographic, use the varying levels of regard towards different demographics as a defining metric for bias in NLG, and analyze the extent to which sentiment scores are a relevant proxy metric for regard. To this end, we collect strategically-generated text from language models and manually annotate the text with both sentiment and regard scores. Additionally, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.01326","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.01326/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.01326","created_at":"2026-07-05T00:14:33.564814+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.01326v2","created_at":"2026-07-05T00:14:33.564814+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.01326","created_at":"2026-07-05T00:14:33.564814+00:00"},{"alias_kind":"pith_short_12","alias_value":"F5HQG4X3AZE3","created_at":"2026-07-05T00:14:33.564814+00:00"},{"alias_kind":"pith_short_16","alias_value":"F5HQG4X3AZE3P6G5","created_at":"2026-07-05T00:14:33.564814+00:00"},{"alias_kind":"pith_short_8","alias_value":"F5HQG4X3","created_at":"2026-07-05T00:14:33.564814+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2211.09085","citing_title":"Galactica: A Large Language Model for Science","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10582","citing_title":"Guaranteed Jailbreaking Defense via Disrupt-and-Rectify Smoothing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07096","citing_title":"Query-efficient model evaluation using cached responses","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05483","citing_title":"Can We Trust a Black-box LLM? LLM Untrustworthy Boundary Detection via Bias-Diffusion and Multi-Agent Reinforcement Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04410","citing_title":"Relative Density Ratio Optimization for Stable and Statistically Consistent Model Alignment","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2005.14165","citing_title":"Language Models are Few-Shot Learners","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21152","citing_title":"Dialect vs Demographics: Quantifying LLM Bias from Implicit Linguistic Signals vs. Explicit User Profiles","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT","json":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT.json","graph_json":"https://pith.science/api/pith-number/F5HQG4X3AZE3P6G5BI26O5ATHT/graph.json","events_json":"https://pith.science/api/pith-number/F5HQG4X3AZE3P6G5BI26O5ATHT/events.json","paper":"https://pith.science/paper/F5HQG4X3"},"agent_actions":{"view_html":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT","download_json":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT.json","view_paper":"https://pith.science/paper/F5HQG4X3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.01326&json=true","fetch_graph":"https://pith.science/api/pith-number/F5HQG4X3AZE3P6G5BI26O5ATHT/graph.json","fetch_events":"https://pith.science/api/pith-number/F5HQG4X3AZE3P6G5BI26O5ATHT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT/action/storage_attestation","attest_author":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT/action/author_attestation","sign_citation":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT/action/citation_signature","submit_replication":"https://pith.science/pith/F5HQG4X3AZE3P6G5BI26O5ATHT/action/replication_record"}},"created_at":"2026-07-05T00:14:33.564814+00:00","updated_at":"2026-07-05T00:14:33.564814+00:00"}