{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XGLYYXN7TQP6KSA66O474QQXPE","short_pith_number":"pith:XGLYYXN7","schema_version":"1.0","canonical_sha256":"b9978c5dbf9c1fe5481ef3b9fe4217793c8d39d52ce4022950c283f3eb81478f","source":{"kind":"arxiv","id":"2105.04054","version":3},"attestation_state":"computed","paper":{"title":"Societal Biases in Language Generation: Progress and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Emily Sheng, Kai-Wei Chang, Nanyun Peng, Premkumar Natarajan","submitted_at":"2021-05-10T00:17:33Z","abstract_excerpt":"Technology for language generation has advanced rapidly, spurred by advancements in pre-training large models on massive amounts of data and the need for intelligent agents to communicate in a natural manner. While techniques can effectively generate fluent text, they can also produce undesirable societal biases that can have a disproportionately negative impact on marginalized populations. Language generation presents unique challenges for biases in terms of direct user interaction and the structure of decoding techniques. To better understand these challenges, we present a survey on societal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.04054","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-05-10T00:17:33Z","cross_cats_sorted":[],"title_canon_sha256":"e25304e81ae3fb07539c03b9cd8a689810abc30a9eb21f33b9e6ccbd846cff58","abstract_canon_sha256":"ccf603da3be572262a827ac199406a7c3e9dfad1effeec029714550f46acb801"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:51:40.369645Z","signature_b64":"eHx2hGzhypaz9pdK0fslpskAH3eGA5INnJK03Xrt3iATsehv8otf7TZj6Gyps0BG5eWFP9hHtPDI8SObd/XdCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9978c5dbf9c1fe5481ef3b9fe4217793c8d39d52ce4022950c283f3eb81478f","last_reissued_at":"2026-07-05T02:51:40.369130Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:51:40.369130Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Societal Biases in Language Generation: Progress and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Emily Sheng, Kai-Wei Chang, Nanyun Peng, Premkumar Natarajan","submitted_at":"2021-05-10T00:17:33Z","abstract_excerpt":"Technology for language generation has advanced rapidly, spurred by advancements in pre-training large models on massive amounts of data and the need for intelligent agents to communicate in a natural manner. While techniques can effectively generate fluent text, they can also produce undesirable societal biases that can have a disproportionately negative impact on marginalized populations. Language generation presents unique challenges for biases in terms of direct user interaction and the structure of decoding techniques. To better understand these challenges, we present a survey on societal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.04054","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.04054/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.04054","created_at":"2026-07-05T02:51:40.369187+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.04054v3","created_at":"2026-07-05T02:51:40.369187+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.04054","created_at":"2026-07-05T02:51:40.369187+00:00"},{"alias_kind":"pith_short_12","alias_value":"XGLYYXN7TQP6","created_at":"2026-07-05T02:51:40.369187+00:00"},{"alias_kind":"pith_short_16","alias_value":"XGLYYXN7TQP6KSA6","created_at":"2026-07-05T02:51:40.369187+00:00"},{"alias_kind":"pith_short_8","alias_value":"XGLYYXN7","created_at":"2026-07-05T02:51:40.369187+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.19444","citing_title":"AI Failures in the Eyes of the Downstream Developer: A First Look at Concerns, Practices, and Challenges","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2211.09085","citing_title":"Galactica: A Large Language Model for Science","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":134,"is_internal_anchor":false},{"citing_arxiv_id":"2112.04359","citing_title":"Ethical and social risks of harm from Language Models","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2204.02311","citing_title":"PaLM: Scaling Language Modeling with Pathways","ref_index":142,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE","json":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE.json","graph_json":"https://pith.science/api/pith-number/XGLYYXN7TQP6KSA66O474QQXPE/graph.json","events_json":"https://pith.science/api/pith-number/XGLYYXN7TQP6KSA66O474QQXPE/events.json","paper":"https://pith.science/paper/XGLYYXN7"},"agent_actions":{"view_html":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE","download_json":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE.json","view_paper":"https://pith.science/paper/XGLYYXN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.04054&json=true","fetch_graph":"https://pith.science/api/pith-number/XGLYYXN7TQP6KSA66O474QQXPE/graph.json","fetch_events":"https://pith.science/api/pith-number/XGLYYXN7TQP6KSA66O474QQXPE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE/action/storage_attestation","attest_author":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE/action/author_attestation","sign_citation":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE/action/citation_signature","submit_replication":"https://pith.science/pith/XGLYYXN7TQP6KSA66O474QQXPE/action/replication_record"}},"created_at":"2026-07-05T02:51:40.369187+00:00","updated_at":"2026-07-05T02:51:40.369187+00:00"}