{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:37CFRFAUUXG7G3HQEMV4XGHY5C","short_pith_number":"pith:37CFRFAU","schema_version":"1.0","canonical_sha256":"dfc4589414a5cdf36cf0232bcb98f8e8920ff2c7b315f0f13c6af22acbf720ef","source":{"kind":"arxiv","id":"2210.11359","version":1},"attestation_state":"computed","paper":{"title":"Data-Efficient Strategies for Expanding Hate Speech Detection into Under-Resourced Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Debora Nozza, Dirk Hovy, Federico Bianchi, Paul R\\\"ottger","submitted_at":"2022-10-20T15:49:00Z","abstract_excerpt":"Hate speech is a global phenomenon, but most hate speech datasets so far focus on English-language content. This hinders the development of more effective hate speech detection models in hundreds of languages spoken by billions across the world. More data is needed, but annotating hateful content is expensive, time-consuming and potentially harmful to annotators. To mitigate these issues, we explore data-efficient strategies for expanding hate speech detection into under-resourced languages. In a series of experiments with mono- and multilingual models across five non-English languages, we fin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.11359","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-20T15:49:00Z","cross_cats_sorted":[],"title_canon_sha256":"6b235b5ab2dc495f0177f101d9e0415ff4cf89acbd09b6fa3a4e5ca1e1cf938d","abstract_canon_sha256":"10607d35f10877378dc5a0973f0d6139ec31eb939aa2af308394cf83b8778dd2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:08:47.229100Z","signature_b64":"Ti4JzfNWOkFUVgWrQeImPZtEvM7JA0XbH09Dt0NtfwWbb0/cjvAGi+1hLJzUWptdCvXiKaSnuZOkOh//8sprDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dfc4589414a5cdf36cf0232bcb98f8e8920ff2c7b315f0f13c6af22acbf720ef","last_reissued_at":"2026-07-05T05:08:47.228690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:08:47.228690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Data-Efficient Strategies for Expanding Hate Speech Detection into Under-Resourced Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Debora Nozza, Dirk Hovy, Federico Bianchi, Paul R\\\"ottger","submitted_at":"2022-10-20T15:49:00Z","abstract_excerpt":"Hate speech is a global phenomenon, but most hate speech datasets so far focus on English-language content. This hinders the development of more effective hate speech detection models in hundreds of languages spoken by billions across the world. More data is needed, but annotating hateful content is expensive, time-consuming and potentially harmful to annotators. To mitigate these issues, we explore data-efficient strategies for expanding hate speech detection into under-resourced languages. In a series of experiments with mono- and multilingual models across five non-English languages, we fin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.11359","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.11359/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.11359","created_at":"2026-07-05T05:08:47.228751+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.11359v1","created_at":"2026-07-05T05:08:47.228751+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.11359","created_at":"2026-07-05T05:08:47.228751+00:00"},{"alias_kind":"pith_short_12","alias_value":"37CFRFAUUXG7","created_at":"2026-07-05T05:08:47.228751+00:00"},{"alias_kind":"pith_short_16","alias_value":"37CFRFAUUXG7G3HQ","created_at":"2026-07-05T05:08:47.228751+00:00"},{"alias_kind":"pith_short_8","alias_value":"37CFRFAU","created_at":"2026-07-05T05:08:47.228751+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.05871","citing_title":"Collaborative Content Moderation in the Fediverse","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C","json":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C.json","graph_json":"https://pith.science/api/pith-number/37CFRFAUUXG7G3HQEMV4XGHY5C/graph.json","events_json":"https://pith.science/api/pith-number/37CFRFAUUXG7G3HQEMV4XGHY5C/events.json","paper":"https://pith.science/paper/37CFRFAU"},"agent_actions":{"view_html":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C","download_json":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C.json","view_paper":"https://pith.science/paper/37CFRFAU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.11359&json=true","fetch_graph":"https://pith.science/api/pith-number/37CFRFAUUXG7G3HQEMV4XGHY5C/graph.json","fetch_events":"https://pith.science/api/pith-number/37CFRFAUUXG7G3HQEMV4XGHY5C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C/action/storage_attestation","attest_author":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C/action/author_attestation","sign_citation":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C/action/citation_signature","submit_replication":"https://pith.science/pith/37CFRFAUUXG7G3HQEMV4XGHY5C/action/replication_record"}},"created_at":"2026-07-05T05:08:47.228751+00:00","updated_at":"2026-07-05T05:08:47.228751+00:00"}