{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YOOBS5AF4YESDYRZEJMW4JWRMQ","short_pith_number":"pith:YOOBS5AF","schema_version":"1.0","canonical_sha256":"c39c197405e60921e23922596e26d16400dfda7f92f2d8f086877aebd89c9f29","source":{"kind":"arxiv","id":"2505.15773","version":1},"attestation_state":"computed","paper":{"title":"ToxicTone: A Mandarin Audio Dataset Annotated for Toxicity and Toxic Utterance Tonality","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Bo-Han Feng, Chien-Feng Liu, Hung-yi Lee, I-Ning Tsai, Jia-Hung Chen, Ming-To Chuang, Pei Xing Kiew, Wenze Ren, Yi-Cheng Lin, Yu-Chen Chen, Yueh-Hsuan Huang, Yu-Xiang Luo","submitted_at":"2025-05-21T17:25:27Z","abstract_excerpt":"Despite extensive research on toxic speech detection in text, a critical gap remains in handling spoken Mandarin audio. The lack of annotated datasets that capture the unique prosodic cues and culturally specific expressions in Mandarin leaves spoken toxicity underexplored. To address this, we introduce ToxicTone -- the largest public dataset of its kind -- featuring detailed annotations that distinguish both forms of toxicity (e.g., profanity, bullying) and sources of toxicity (e.g., anger, sarcasm, dismissiveness). Our data, sourced from diverse real-world audio and organized into 13 topical"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15773","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"eess.AS","submitted_at":"2025-05-21T17:25:27Z","cross_cats_sorted":["cs.CL","cs.SD"],"title_canon_sha256":"1f940b6d5dd5885f78809420c497b708ebf5e34d664f404bda0142ed234c189e","abstract_canon_sha256":"9f05266c892152d7167fb34c62aa406fc479a45a501514c53808d352941a91a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:57.711032Z","signature_b64":"NnXo4wVkDVQqRM+Vvbnjj+5pfftBZqYabLDj7PZ9jBpGPbZXdsEcerp1VKlgdx91uSj8gDefqik5mnaFbsjoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c39c197405e60921e23922596e26d16400dfda7f92f2d8f086877aebd89c9f29","last_reissued_at":"2026-07-05T11:06:57.710568Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:57.710568Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ToxicTone: A Mandarin Audio Dataset Annotated for Toxicity and Toxic Utterance Tonality","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Bo-Han Feng, Chien-Feng Liu, Hung-yi Lee, I-Ning Tsai, Jia-Hung Chen, Ming-To Chuang, Pei Xing Kiew, Wenze Ren, Yi-Cheng Lin, Yu-Chen Chen, Yueh-Hsuan Huang, Yu-Xiang Luo","submitted_at":"2025-05-21T17:25:27Z","abstract_excerpt":"Despite extensive research on toxic speech detection in text, a critical gap remains in handling spoken Mandarin audio. The lack of annotated datasets that capture the unique prosodic cues and culturally specific expressions in Mandarin leaves spoken toxicity underexplored. To address this, we introduce ToxicTone -- the largest public dataset of its kind -- featuring detailed annotations that distinguish both forms of toxicity (e.g., profanity, bullying) and sources of toxicity (e.g., anger, sarcasm, dismissiveness). Our data, sourced from diverse real-world audio and organized into 13 topical"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15773","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15773/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15773","created_at":"2026-07-05T11:06:57.710629+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15773v1","created_at":"2026-07-05T11:06:57.710629+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15773","created_at":"2026-07-05T11:06:57.710629+00:00"},{"alias_kind":"pith_short_12","alias_value":"YOOBS5AF4YES","created_at":"2026-07-05T11:06:57.710629+00:00"},{"alias_kind":"pith_short_16","alias_value":"YOOBS5AF4YESDYRZ","created_at":"2026-07-05T11:06:57.710629+00:00"},{"alias_kind":"pith_short_8","alias_value":"YOOBS5AF","created_at":"2026-07-05T11:06:57.710629+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.15773","citing_title":"ToxicTone: A Mandarin Audio Dataset Annotated for Toxicity and Toxic Utterance Tonality","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ","json":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ.json","graph_json":"https://pith.science/api/pith-number/YOOBS5AF4YESDYRZEJMW4JWRMQ/graph.json","events_json":"https://pith.science/api/pith-number/YOOBS5AF4YESDYRZEJMW4JWRMQ/events.json","paper":"https://pith.science/paper/YOOBS5AF"},"agent_actions":{"view_html":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ","download_json":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ.json","view_paper":"https://pith.science/paper/YOOBS5AF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15773&json=true","fetch_graph":"https://pith.science/api/pith-number/YOOBS5AF4YESDYRZEJMW4JWRMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/YOOBS5AF4YESDYRZEJMW4JWRMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ/action/storage_attestation","attest_author":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ/action/author_attestation","sign_citation":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ/action/citation_signature","submit_replication":"https://pith.science/pith/YOOBS5AF4YESDYRZEJMW4JWRMQ/action/replication_record"}},"created_at":"2026-07-05T11:06:57.710629+00:00","updated_at":"2026-07-05T11:06:57.710629+00:00"}