{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NKSWGSTKE2EBZDAUOEY4QGX67U","short_pith_number":"pith:NKSWGSTK","schema_version":"1.0","canonical_sha256":"6aa5634a6a26881c8c147131c81afefd3a56ae82ebd3016bc851b74e17f55a4c","source":{"kind":"arxiv","id":"2509.03959","version":2},"attestation_state":"computed","paper":{"title":"WenetSpeech-Yue: A Large-scale Cantonese Speech Corpus with Multi-dimensional Annotation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SD","authors_text":"Binbin Zhang, Chengyou Wang, Hongfei Xue, Hongjie Chen, Hui Bu, Jian Kang, Jie Li, Lei Xie, Longhao Li, Ruibin Yuan, Shuiyuan Wang, Tianlun Zuo, Wei Xue, Xin Xu, Yuhang Dai, Zhao Guo, Ziya Zhou, Ziyu Zhang","submitted_at":"2025-09-04T07:36:43Z","abstract_excerpt":"The development of speech understanding and generation has been significantly accelerated by the availability of large-scale, high-quality speech datasets. Among these, ASR and TTS are regarded as the most established and fundamental tasks. However, for Cantonese (Yue Chinese), spoken by approximately 84.9 million native speakers worldwide, limited annotated resources have hindered progress and resulted in suboptimal ASR and TTS performance. To address this challenge, we propose WenetSpeech-Pipe, an integrated pipeline for building large-scale speech corpus with multi-dimensional annotation ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.03959","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-09-04T07:36:43Z","cross_cats_sorted":[],"title_canon_sha256":"4ce526c97dfde1a5ec5c76c9471b173b43c0e82122b67dcff4560bc0025d1622","abstract_canon_sha256":"eb6d9e379193f123bb767a0544ce0c41026dbfd6d2b07b80f7a9743f221e19f3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:22.835810Z","signature_b64":"8pqDL6aEnPqTKMWh0yzc6unLX3aeNjAm5ZpA/SEqWPVWYV9IKLF0tgkI1s6+N9Sd2vUjD8xDdXbwLtfRMBAoAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6aa5634a6a26881c8c147131c81afefd3a56ae82ebd3016bc851b74e17f55a4c","last_reissued_at":"2026-07-05T12:05:22.835289Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:22.835289Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WenetSpeech-Yue: A Large-scale Cantonese Speech Corpus with Multi-dimensional Annotation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SD","authors_text":"Binbin Zhang, Chengyou Wang, Hongfei Xue, Hongjie Chen, Hui Bu, Jian Kang, Jie Li, Lei Xie, Longhao Li, Ruibin Yuan, Shuiyuan Wang, Tianlun Zuo, Wei Xue, Xin Xu, Yuhang Dai, Zhao Guo, Ziya Zhou, Ziyu Zhang","submitted_at":"2025-09-04T07:36:43Z","abstract_excerpt":"The development of speech understanding and generation has been significantly accelerated by the availability of large-scale, high-quality speech datasets. Among these, ASR and TTS are regarded as the most established and fundamental tasks. However, for Cantonese (Yue Chinese), spoken by approximately 84.9 million native speakers worldwide, limited annotated resources have hindered progress and resulted in suboptimal ASR and TTS performance. To address this challenge, we propose WenetSpeech-Pipe, an integrated pipeline for building large-scale speech corpus with multi-dimensional annotation ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.03959","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.03959/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.03959","created_at":"2026-07-05T12:05:22.835375+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.03959v2","created_at":"2026-07-05T12:05:22.835375+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.03959","created_at":"2026-07-05T12:05:22.835375+00:00"},{"alias_kind":"pith_short_12","alias_value":"NKSWGSTKE2EB","created_at":"2026-07-05T12:05:22.835375+00:00"},{"alias_kind":"pith_short_16","alias_value":"NKSWGSTKE2EBZDAU","created_at":"2026-07-05T12:05:22.835375+00:00"},{"alias_kind":"pith_short_8","alias_value":"NKSWGSTK","created_at":"2026-07-05T12:05:22.835375+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18105","citing_title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01016","citing_title":"PolySpeech-100: A Large-Scale Benchmark for Speech Understanding Across 100+ Languages and Dialects","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17846","citing_title":"UrduSpeech: A 156-Hour Urdu Speech Corpus with 12-Dimension Paralinguistic Annotations","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00688","citing_title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18105","citing_title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U","json":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U.json","graph_json":"https://pith.science/api/pith-number/NKSWGSTKE2EBZDAUOEY4QGX67U/graph.json","events_json":"https://pith.science/api/pith-number/NKSWGSTKE2EBZDAUOEY4QGX67U/events.json","paper":"https://pith.science/paper/NKSWGSTK"},"agent_actions":{"view_html":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U","download_json":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U.json","view_paper":"https://pith.science/paper/NKSWGSTK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.03959&json=true","fetch_graph":"https://pith.science/api/pith-number/NKSWGSTKE2EBZDAUOEY4QGX67U/graph.json","fetch_events":"https://pith.science/api/pith-number/NKSWGSTKE2EBZDAUOEY4QGX67U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U/action/storage_attestation","attest_author":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U/action/author_attestation","sign_citation":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U/action/citation_signature","submit_replication":"https://pith.science/pith/NKSWGSTKE2EBZDAUOEY4QGX67U/action/replication_record"}},"created_at":"2026-07-05T12:05:22.835375+00:00","updated_at":"2026-07-05T12:05:22.835375+00:00"}