{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JBAZTXOFBJMD44WXOY23HEHDEF","short_pith_number":"pith:JBAZTXOF","schema_version":"1.0","canonical_sha256":"484199ddc50a583e72d77635b390e32155f5613e74007330b6843ced772e73b1","source":{"kind":"arxiv","id":"2404.00303","version":1},"attestation_state":"computed","paper":{"title":"A Comprehensive Study on NLP Data Augmentation for Hate Speech Detection: Legacy Methods, BERT, and LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Djamila Romaissa Beddia, Jhuma kabir Mim, Md Saroar Jahan, Mourad Oussalah, Nabil Arhab","submitted_at":"2024-03-30T09:55:58Z","abstract_excerpt":"The surge of interest in data augmentation within the realm of NLP has been driven by the need to address challenges posed by hate speech domains, the dynamic nature of social media vocabulary, and the demands for large-scale neural networks requiring extensive training data. However, the prevalent use of lexical substitution in data augmentation has raised concerns, as it may inadvertently alter the intended meaning, thereby impacting the efficacy of supervised machine learning models. In pursuit of suitable data augmentation methods, this study explores both established legacy approaches and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00303","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-30T09:55:58Z","cross_cats_sorted":[],"title_canon_sha256":"09df37c46a54c2eb232f66fa435b0de21cf506b616a1015ecb15c7df92daeda8","abstract_canon_sha256":"4f0d573e64a28aec11129b690a0290e3243b136e2b6e2db444c7596fa9a278b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:31.161284Z","signature_b64":"0d5s521JTeYHdYzHFfMqnQunOCbfIFPu3SGoO+Np1l/wSSjJbwNe8Bbx+0CYvyUCzPDmEmLX3UbmVhzgv9eIBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"484199ddc50a583e72d77635b390e32155f5613e74007330b6843ced772e73b1","last_reissued_at":"2026-07-05T08:02:31.160852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:31.160852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Study on NLP Data Augmentation for Hate Speech Detection: Legacy Methods, BERT, and LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Djamila Romaissa Beddia, Jhuma kabir Mim, Md Saroar Jahan, Mourad Oussalah, Nabil Arhab","submitted_at":"2024-03-30T09:55:58Z","abstract_excerpt":"The surge of interest in data augmentation within the realm of NLP has been driven by the need to address challenges posed by hate speech domains, the dynamic nature of social media vocabulary, and the demands for large-scale neural networks requiring extensive training data. However, the prevalent use of lexical substitution in data augmentation has raised concerns, as it may inadvertently alter the intended meaning, thereby impacting the efficacy of supervised machine learning models. In pursuit of suitable data augmentation methods, this study explores both established legacy approaches and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00303","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00303/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00303","created_at":"2026-07-05T08:02:31.160910+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00303v1","created_at":"2026-07-05T08:02:31.160910+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00303","created_at":"2026-07-05T08:02:31.160910+00:00"},{"alias_kind":"pith_short_12","alias_value":"JBAZTXOFBJMD","created_at":"2026-07-05T08:02:31.160910+00:00"},{"alias_kind":"pith_short_16","alias_value":"JBAZTXOFBJMD44WX","created_at":"2026-07-05T08:02:31.160910+00:00"},{"alias_kind":"pith_short_8","alias_value":"JBAZTXOF","created_at":"2026-07-05T08:02:31.160910+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.01256","citing_title":"Digital Guardians: Can GPT-4, Perspective API, and Moderation API reliably detect hate speech in reader comments of German online newspapers?","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF","json":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF.json","graph_json":"https://pith.science/api/pith-number/JBAZTXOFBJMD44WXOY23HEHDEF/graph.json","events_json":"https://pith.science/api/pith-number/JBAZTXOFBJMD44WXOY23HEHDEF/events.json","paper":"https://pith.science/paper/JBAZTXOF"},"agent_actions":{"view_html":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF","download_json":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF.json","view_paper":"https://pith.science/paper/JBAZTXOF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00303&json=true","fetch_graph":"https://pith.science/api/pith-number/JBAZTXOFBJMD44WXOY23HEHDEF/graph.json","fetch_events":"https://pith.science/api/pith-number/JBAZTXOFBJMD44WXOY23HEHDEF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF/action/storage_attestation","attest_author":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF/action/author_attestation","sign_citation":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF/action/citation_signature","submit_replication":"https://pith.science/pith/JBAZTXOFBJMD44WXOY23HEHDEF/action/replication_record"}},"created_at":"2026-07-05T08:02:31.160910+00:00","updated_at":"2026-07-05T08:02:31.160910+00:00"}