{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZLAD45GB5D5WULZF7UNUDTCORX","short_pith_number":"pith:ZLAD45GB","schema_version":"1.0","canonical_sha256":"cac03e74c1e8fb6a2f25fd1b41cc4e8dd81e155f604eede547b18b6133704f4e","source":{"kind":"arxiv","id":"2210.03479","version":1},"attestation_state":"computed","paper":{"title":"Hate Speech and Offensive Language Detection in Bengali","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Animesh Mukherjee, Mithun Das, Punyajoy Saha, Somnath Banerjee","submitted_at":"2022-10-07T12:06:04Z","abstract_excerpt":"Social media often serves as a breeding ground for various hateful and offensive content. Identifying such content on social media is crucial due to its impact on the race, gender, or religion in an unprejudiced society. However, while there is extensive research in hate speech detection in English, there is a gap in hateful content detection in low-resource languages like Bengali. Besides, a current trend on social media is the use of Romanized Bengali for regular interactions. To overcome the existing research's limitations, in this study, we develop an annotated dataset of 10K Bengali posts"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.03479","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-07T12:06:04Z","cross_cats_sorted":[],"title_canon_sha256":"b9219302cefb2caa6bd9626564bd5834e9c962245c253df3f5bf4cd8c3d448a8","abstract_canon_sha256":"14ade8c04af9517eef67709c4d5b25a9ffda33377a7a28d88b4d15b1bf89129b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:20.671057Z","signature_b64":"lmgzgkV5wL+ET3VmxzbDIX75Xg0Z0XH8lS50p3/EvVLnnrG+C3wCU2eTycANOiuD42Ge2etMUHwj48c94LKoCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cac03e74c1e8fb6a2f25fd1b41cc4e8dd81e155f604eede547b18b6133704f4e","last_reissued_at":"2026-07-05T05:04:20.670676Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:20.670676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hate Speech and Offensive Language Detection in Bengali","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Animesh Mukherjee, Mithun Das, Punyajoy Saha, Somnath Banerjee","submitted_at":"2022-10-07T12:06:04Z","abstract_excerpt":"Social media often serves as a breeding ground for various hateful and offensive content. Identifying such content on social media is crucial due to its impact on the race, gender, or religion in an unprejudiced society. However, while there is extensive research in hate speech detection in English, there is a gap in hateful content detection in low-resource languages like Bengali. Besides, a current trend on social media is the use of Romanized Bengali for regular interactions. To overcome the existing research's limitations, in this study, we develop an annotated dataset of 10K Bengali posts"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.03479","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.03479/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.03479","created_at":"2026-07-05T05:04:20.670731+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.03479v1","created_at":"2026-07-05T05:04:20.670731+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.03479","created_at":"2026-07-05T05:04:20.670731+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZLAD45GB5D5W","created_at":"2026-07-05T05:04:20.670731+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZLAD45GB5D5WULZF","created_at":"2026-07-05T05:04:20.670731+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZLAD45GB","created_at":"2026-07-05T05:04:20.670731+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.16183","citing_title":"BIDWESH: A Bangla Regional Based Hate Speech Detection Dataset","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX","json":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX.json","graph_json":"https://pith.science/api/pith-number/ZLAD45GB5D5WULZF7UNUDTCORX/graph.json","events_json":"https://pith.science/api/pith-number/ZLAD45GB5D5WULZF7UNUDTCORX/events.json","paper":"https://pith.science/paper/ZLAD45GB"},"agent_actions":{"view_html":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX","download_json":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX.json","view_paper":"https://pith.science/paper/ZLAD45GB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.03479&json=true","fetch_graph":"https://pith.science/api/pith-number/ZLAD45GB5D5WULZF7UNUDTCORX/graph.json","fetch_events":"https://pith.science/api/pith-number/ZLAD45GB5D5WULZF7UNUDTCORX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX/action/storage_attestation","attest_author":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX/action/author_attestation","sign_citation":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX/action/citation_signature","submit_replication":"https://pith.science/pith/ZLAD45GB5D5WULZF7UNUDTCORX/action/replication_record"}},"created_at":"2026-07-05T05:04:20.670731+00:00","updated_at":"2026-07-05T05:04:20.670731+00:00"}