{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ILWNGS73Q5VT5B5B64XHCPSDFS","short_pith_number":"pith:ILWNGS73","schema_version":"1.0","canonical_sha256":"42ecd34bfb876b3e87a1f72e713e432cb8d223e96cec562b9273be436f94c79b","source":{"kind":"arxiv","id":"2506.09992","version":2},"attestation_state":"computed","paper":{"title":"Large Language Models for Toxic Language Detection in Low-Resource Balkan Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Amela Kadric Muminovic, Amel Muminovic","submitted_at":"2025-06-11T17:59:33Z","abstract_excerpt":"Online toxic language causes real harm, especially in regions with limited moderation tools. In this study, we evaluate how large language models handle toxic comments in Serbian, Croatian, and Bosnian, languages with limited labeled data. We built and manually labeled a dataset of 4,500 YouTube and TikTok comments drawn from videos across diverse categories, including music, politics, sports, modeling, influencer content, discussions of sexism, and general topics. Four models (GPT-3.5 Turbo, GPT-4.1, Gemini 1.5 Pro, and Claude 3 Opus) were tested in two modes: zero-shot and context-augmented."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09992","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-11T17:59:33Z","cross_cats_sorted":[],"title_canon_sha256":"43ce904ff92926bba2aa43698feae3cbdbec278eb1998100833432d1e77e03e2","abstract_canon_sha256":"7a6c2dc934341aafd0c6cac77cfdc5d02ff4241ae2b54c04ec26b3d0389ffd4b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:56.702336Z","signature_b64":"3lwIjmEAPynHs+9o3DcfsmWni6W3fIhFMHoHr/IY1ZnHek/yC+J+oKAlScho6gTNM80iXoVcEGxJ1lk0ae4YBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42ecd34bfb876b3e87a1f72e713e432cb8d223e96cec562b9273be436f94c79b","last_reissued_at":"2026-07-05T11:20:56.701863Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:56.701863Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Toxic Language Detection in Low-Resource Balkan Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Amela Kadric Muminovic, Amel Muminovic","submitted_at":"2025-06-11T17:59:33Z","abstract_excerpt":"Online toxic language causes real harm, especially in regions with limited moderation tools. In this study, we evaluate how large language models handle toxic comments in Serbian, Croatian, and Bosnian, languages with limited labeled data. We built and manually labeled a dataset of 4,500 YouTube and TikTok comments drawn from videos across diverse categories, including music, politics, sports, modeling, influencer content, discussions of sexism, and general topics. Four models (GPT-3.5 Turbo, GPT-4.1, Gemini 1.5 Pro, and Claude 3 Opus) were tested in two modes: zero-shot and context-augmented."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09992","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09992/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09992","created_at":"2026-07-05T11:20:56.701918+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09992v2","created_at":"2026-07-05T11:20:56.701918+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09992","created_at":"2026-07-05T11:20:56.701918+00:00"},{"alias_kind":"pith_short_12","alias_value":"ILWNGS73Q5VT","created_at":"2026-07-05T11:20:56.701918+00:00"},{"alias_kind":"pith_short_16","alias_value":"ILWNGS73Q5VT5B5B","created_at":"2026-07-05T11:20:56.701918+00:00"},{"alias_kind":"pith_short_8","alias_value":"ILWNGS73","created_at":"2026-07-05T11:20:56.701918+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS","json":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS.json","graph_json":"https://pith.science/api/pith-number/ILWNGS73Q5VT5B5B64XHCPSDFS/graph.json","events_json":"https://pith.science/api/pith-number/ILWNGS73Q5VT5B5B64XHCPSDFS/events.json","paper":"https://pith.science/paper/ILWNGS73"},"agent_actions":{"view_html":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS","download_json":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS.json","view_paper":"https://pith.science/paper/ILWNGS73","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09992&json=true","fetch_graph":"https://pith.science/api/pith-number/ILWNGS73Q5VT5B5B64XHCPSDFS/graph.json","fetch_events":"https://pith.science/api/pith-number/ILWNGS73Q5VT5B5B64XHCPSDFS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS/action/storage_attestation","attest_author":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS/action/author_attestation","sign_citation":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS/action/citation_signature","submit_replication":"https://pith.science/pith/ILWNGS73Q5VT5B5B64XHCPSDFS/action/replication_record"}},"created_at":"2026-07-05T11:20:56.701918+00:00","updated_at":"2026-07-05T11:20:56.701918+00:00"}