{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:LKAFLM47JWUBH2BI7FOBKSHDWC","short_pith_number":"pith:LKAFLM47","schema_version":"1.0","canonical_sha256":"5a8055b39f4da813e828f95c1548e3b0bbe026bea83091458e6beb9c524ed6b2","source":{"kind":"arxiv","id":"2010.12472","version":2},"attestation_state":"computed","paper":{"title":"HateBERT: Retraining BERT for Abusive Language Detection in English","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jelena Mitrovi\\'c, Michael Granitzer, Tommaso Caselli, Valerio Basile","submitted_at":"2020-10-23T15:14:14Z","abstract_excerpt":"In this paper, we introduce HateBERT, a re-trained BERT model for abusive language detection in English. The model was trained on RAL-E, a large-scale dataset of Reddit comments in English from communities banned for being offensive, abusive, or hateful that we have collected and made available to the public. We present the results of a detailed comparison between a general pre-trained language model and the abuse-inclined version obtained by retraining with posts from the banned communities on three English datasets for offensive, abusive language and hate speech detection tasks. In all datas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.12472","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2020-10-23T15:14:14Z","cross_cats_sorted":[],"title_canon_sha256":"eb4bc6d3961f2ca72b949bd9bd2a2090b27fcdf787a4a469562cfe67253801eb","abstract_canon_sha256":"247820f899d6646a41f2140dfadcbc4d599b61fc56242bf20bd7893072cb9ca3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:12:43.492826Z","signature_b64":"zp/4nYfI6e6KudDnf+AlprgKLOck0+jd9Yc/AuW0wiAo8VyMs7qEJDyIgdkj8fmD7qRw8hPMesB4R8DI/DXkCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a8055b39f4da813e828f95c1548e3b0bbe026bea83091458e6beb9c524ed6b2","last_reissued_at":"2026-07-05T02:12:43.492354Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:12:43.492354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HateBERT: Retraining BERT for Abusive Language Detection in English","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jelena Mitrovi\\'c, Michael Granitzer, Tommaso Caselli, Valerio Basile","submitted_at":"2020-10-23T15:14:14Z","abstract_excerpt":"In this paper, we introduce HateBERT, a re-trained BERT model for abusive language detection in English. The model was trained on RAL-E, a large-scale dataset of Reddit comments in English from communities banned for being offensive, abusive, or hateful that we have collected and made available to the public. We present the results of a detailed comparison between a general pre-trained language model and the abuse-inclined version obtained by retraining with posts from the banned communities on three English datasets for offensive, abusive language and hate speech detection tasks. In all datas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.12472","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.12472/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.12472","created_at":"2026-07-05T02:12:43.492413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.12472v2","created_at":"2026-07-05T02:12:43.492413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.12472","created_at":"2026-07-05T02:12:43.492413+00:00"},{"alias_kind":"pith_short_12","alias_value":"LKAFLM47JWUB","created_at":"2026-07-05T02:12:43.492413+00:00"},{"alias_kind":"pith_short_16","alias_value":"LKAFLM47JWUBH2BI","created_at":"2026-07-05T02:12:43.492413+00:00"},{"alias_kind":"pith_short_8","alias_value":"LKAFLM47","created_at":"2026-07-05T02:12:43.492413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.20654","citing_title":"Chinese Cyberbullying Detection: Dataset, Method, and Validation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2306.02707","citing_title":"Orca: Progressive Learning from Complex Explanation Traces of GPT-4","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10639","citing_title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC","json":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC.json","graph_json":"https://pith.science/api/pith-number/LKAFLM47JWUBH2BI7FOBKSHDWC/graph.json","events_json":"https://pith.science/api/pith-number/LKAFLM47JWUBH2BI7FOBKSHDWC/events.json","paper":"https://pith.science/paper/LKAFLM47"},"agent_actions":{"view_html":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC","download_json":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC.json","view_paper":"https://pith.science/paper/LKAFLM47","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.12472&json=true","fetch_graph":"https://pith.science/api/pith-number/LKAFLM47JWUBH2BI7FOBKSHDWC/graph.json","fetch_events":"https://pith.science/api/pith-number/LKAFLM47JWUBH2BI7FOBKSHDWC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC/action/storage_attestation","attest_author":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC/action/author_attestation","sign_citation":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC/action/citation_signature","submit_replication":"https://pith.science/pith/LKAFLM47JWUBH2BI7FOBKSHDWC/action/replication_record"}},"created_at":"2026-07-05T02:12:43.492413+00:00","updated_at":"2026-07-05T02:12:43.492413+00:00"}