{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TNZ3RZB3E665LFWT42SHINKFLN","short_pith_number":"pith:TNZ3RZB3","schema_version":"1.0","canonical_sha256":"9b73b8e43b27bdd596d3e6a47435455b48228f153b77f8a781fd44645533633e","source":{"kind":"arxiv","id":"2306.02294","version":1},"attestation_state":"computed","paper":{"title":"Exposing Bias in Online Communities through Large-Scale Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Celine Wald, Lukas Pfahler","submitted_at":"2023-06-04T08:09:26Z","abstract_excerpt":"Progress in natural language generation research has been shaped by the ever-growing size of language models. While large language models pre-trained on web data can generate human-sounding text, they also reproduce social biases and contribute to the propagation of harmful stereotypes. This work utilises the flaw of bias in language models to explore the biases of six different online communities. In order to get an insight into the communities' viewpoints, we fine-tune GPT-Neo 1.3B with six social media datasets. The bias of the resulting models is evaluated by prompting the models with diff"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.02294","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-04T08:09:26Z","cross_cats_sorted":["cs.CY","cs.LG"],"title_canon_sha256":"aefe50b4444eaf9c60cbe9e4890ad287da930149e42e86e57b5581e92fa21e80","abstract_canon_sha256":"1e69278ead0571a47fdbcf4a686b4f482cdc811bb0b3900d426ffb29c4cae3e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:15.362898Z","signature_b64":"rT5E/G/T0w9FJbW7iA9KtWKOYqmpUAeYP9nUDFKfJ3FPgHBzlVp3/FWR1Be5bi7PpD/27+kwwK7dhdPXaWOiAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b73b8e43b27bdd596d3e6a47435455b48228f153b77f8a781fd44645533633e","last_reissued_at":"2026-07-05T06:17:15.362375Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:15.362375Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exposing Bias in Online Communities through Large-Scale Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Celine Wald, Lukas Pfahler","submitted_at":"2023-06-04T08:09:26Z","abstract_excerpt":"Progress in natural language generation research has been shaped by the ever-growing size of language models. While large language models pre-trained on web data can generate human-sounding text, they also reproduce social biases and contribute to the propagation of harmful stereotypes. This work utilises the flaw of bias in language models to explore the biases of six different online communities. In order to get an insight into the communities' viewpoints, we fine-tune GPT-Neo 1.3B with six social media datasets. The bias of the resulting models is evaluated by prompting the models with diff"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.02294","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.02294/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.02294","created_at":"2026-07-05T06:17:15.362436+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.02294v1","created_at":"2026-07-05T06:17:15.362436+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.02294","created_at":"2026-07-05T06:17:15.362436+00:00"},{"alias_kind":"pith_short_12","alias_value":"TNZ3RZB3E665","created_at":"2026-07-05T06:17:15.362436+00:00"},{"alias_kind":"pith_short_16","alias_value":"TNZ3RZB3E665LFWT","created_at":"2026-07-05T06:17:15.362436+00:00"},{"alias_kind":"pith_short_8","alias_value":"TNZ3RZB3","created_at":"2026-07-05T06:17:15.362436+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN","json":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN.json","graph_json":"https://pith.science/api/pith-number/TNZ3RZB3E665LFWT42SHINKFLN/graph.json","events_json":"https://pith.science/api/pith-number/TNZ3RZB3E665LFWT42SHINKFLN/events.json","paper":"https://pith.science/paper/TNZ3RZB3"},"agent_actions":{"view_html":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN","download_json":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN.json","view_paper":"https://pith.science/paper/TNZ3RZB3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.02294&json=true","fetch_graph":"https://pith.science/api/pith-number/TNZ3RZB3E665LFWT42SHINKFLN/graph.json","fetch_events":"https://pith.science/api/pith-number/TNZ3RZB3E665LFWT42SHINKFLN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN/action/storage_attestation","attest_author":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN/action/author_attestation","sign_citation":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN/action/citation_signature","submit_replication":"https://pith.science/pith/TNZ3RZB3E665LFWT42SHINKFLN/action/replication_record"}},"created_at":"2026-07-05T06:17:15.362436+00:00","updated_at":"2026-07-05T06:17:15.362436+00:00"}