{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TZGOW6IUTF7X4MVVEW7OFATF7W","short_pith_number":"pith:TZGOW6IU","schema_version":"1.0","canonical_sha256":"9e4ceb7914997f7e32b525bee28265fdaca53f73490ae1de2eaaed9ed3db163e","source":{"kind":"arxiv","id":"2506.00636","version":1},"attestation_state":"computed","paper":{"title":"ViToSA: Audio-Based Toxic Spans Detection on Vietnamese Speech Utterances","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Huy Ba Do, Luan Thanh Nguyen, Vy Le-Phuong Huynh","submitted_at":"2025-05-31T17:01:18Z","abstract_excerpt":"Toxic speech on online platforms is a growing concern, impacting user experience and online safety. While text-based toxicity detection is well-studied, audio-based approaches remain underexplored, especially for low-resource languages like Vietnamese. This paper introduces ViToSA (Vietnamese Toxic Spans Audio), the first dataset for toxic spans detection in Vietnamese speech, comprising 11,000 audio samples (25 hours) with accurate human-annotated transcripts. We propose a pipeline that combines ASR and toxic spans detection for fine-grained identification of toxic content. Our experiments sh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00636","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-31T17:01:18Z","cross_cats_sorted":[],"title_canon_sha256":"e960981f03fa54f7b4ffa8d70aa4ed6d0148dff933d3d4f0962656ecd0ee0ca4","abstract_canon_sha256":"a9247a93ec8ee7b07cd4ba8f5fe21e10595f614e16a33e8906a36d2b32e8dc99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:37.983576Z","signature_b64":"nyAdGB4Q6C9FSQoG+c4Pp/okPKahrnLgLh4WVFh786gKo90T5ewxDoiIpH1jZrSte59PsSEJED/Fpok+YmesAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e4ceb7914997f7e32b525bee28265fdaca53f73490ae1de2eaaed9ed3db163e","last_reissued_at":"2026-07-05T11:13:37.983146Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:37.983146Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViToSA: Audio-Based Toxic Spans Detection on Vietnamese Speech Utterances","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Huy Ba Do, Luan Thanh Nguyen, Vy Le-Phuong Huynh","submitted_at":"2025-05-31T17:01:18Z","abstract_excerpt":"Toxic speech on online platforms is a growing concern, impacting user experience and online safety. While text-based toxicity detection is well-studied, audio-based approaches remain underexplored, especially for low-resource languages like Vietnamese. This paper introduces ViToSA (Vietnamese Toxic Spans Audio), the first dataset for toxic spans detection in Vietnamese speech, comprising 11,000 audio samples (25 hours) with accurate human-annotated transcripts. We propose a pipeline that combines ASR and toxic spans detection for fine-grained identification of toxic content. Our experiments sh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00636","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00636/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00636","created_at":"2026-07-05T11:13:37.983201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00636v1","created_at":"2026-07-05T11:13:37.983201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00636","created_at":"2026-07-05T11:13:37.983201+00:00"},{"alias_kind":"pith_short_12","alias_value":"TZGOW6IUTF7X","created_at":"2026-07-05T11:13:37.983201+00:00"},{"alias_kind":"pith_short_16","alias_value":"TZGOW6IUTF7X4MVV","created_at":"2026-07-05T11:13:37.983201+00:00"},{"alias_kind":"pith_short_8","alias_value":"TZGOW6IU","created_at":"2026-07-05T11:13:37.983201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00636","citing_title":"ViToSA: Audio-Based Toxic Spans Detection on Vietnamese Speech Utterances","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W","json":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W.json","graph_json":"https://pith.science/api/pith-number/TZGOW6IUTF7X4MVVEW7OFATF7W/graph.json","events_json":"https://pith.science/api/pith-number/TZGOW6IUTF7X4MVVEW7OFATF7W/events.json","paper":"https://pith.science/paper/TZGOW6IU"},"agent_actions":{"view_html":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W","download_json":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W.json","view_paper":"https://pith.science/paper/TZGOW6IU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00636&json=true","fetch_graph":"https://pith.science/api/pith-number/TZGOW6IUTF7X4MVVEW7OFATF7W/graph.json","fetch_events":"https://pith.science/api/pith-number/TZGOW6IUTF7X4MVVEW7OFATF7W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W/action/storage_attestation","attest_author":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W/action/author_attestation","sign_citation":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W/action/citation_signature","submit_replication":"https://pith.science/pith/TZGOW6IUTF7X4MVVEW7OFATF7W/action/replication_record"}},"created_at":"2026-07-05T11:13:37.983201+00:00","updated_at":"2026-07-05T11:13:37.983201+00:00"}