{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G34LGQL4CHZANZJ6MZ6OFGDKHX","short_pith_number":"pith:G34LGQL4","schema_version":"1.0","canonical_sha256":"36f8b3417c11f206e53e667ce2986a3dc68a409ee76c1c3f1390b2e1ea382402","source":{"kind":"arxiv","id":"2409.18584","version":3},"attestation_state":"computed","paper":{"title":"ChildMandarin: A Comprehensive Mandarin Speech Dataset for Young Children Aged 3-5","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Aobo Kong, Cheng Liu, Haoqin Sun, Hui Wang, Jiabei He, Jiaming Zhou, Shiwan Zhao, Shiyao Wang, Xi Yang, Yequan Wang, Yonghua Lin, Yong Qin, Yujie Guo","submitted_at":"2024-09-27T09:42:27Z","abstract_excerpt":"Automatic speech recognition (ASR) systems have advanced significantly with models like Whisper, Conformer, and self-supervised frameworks such as Wav2vec 2.0 and HuBERT. However, developing robust ASR models for young children's speech remains challenging due to differences in pronunciation, tone, and pace compared to adult speech. In this paper, we introduce a new Mandarin speech dataset focused on children aged 3 to 5, addressing the scarcity of resources in this area. The dataset comprises 41.25 hours of speech with carefully crafted manual transcriptions, collected from 397 speakers acros"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.18584","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-09-27T09:42:27Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"bc46770685f204e62c6991a9202361a6f288ea43b23be1558c8b0594cbfa39ac","abstract_canon_sha256":"928c953d4bcb7fd5be5221ca4736ab14a84060d31d63c9d425f53e0f7ec7dc59"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:34:39.781385Z","signature_b64":"OIqE/5sQAxYmMs+Oi7vGmw9QEUDv0DXFg3bfdaz2jh7J7bSJAPizPtduBfmcHNdAyIbqb9f/ySALm85+JTIiCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36f8b3417c11f206e53e667ce2986a3dc68a409ee76c1c3f1390b2e1ea382402","last_reissued_at":"2026-07-05T10:34:39.780719Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:34:39.780719Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChildMandarin: A Comprehensive Mandarin Speech Dataset for Young Children Aged 3-5","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.SD","authors_text":"Aobo Kong, Cheng Liu, Haoqin Sun, Hui Wang, Jiabei He, Jiaming Zhou, Shiwan Zhao, Shiyao Wang, Xi Yang, Yequan Wang, Yonghua Lin, Yong Qin, Yujie Guo","submitted_at":"2024-09-27T09:42:27Z","abstract_excerpt":"Automatic speech recognition (ASR) systems have advanced significantly with models like Whisper, Conformer, and self-supervised frameworks such as Wav2vec 2.0 and HuBERT. However, developing robust ASR models for young children's speech remains challenging due to differences in pronunciation, tone, and pace compared to adult speech. In this paper, we introduce a new Mandarin speech dataset focused on children aged 3 to 5, addressing the scarcity of resources in this area. The dataset comprises 41.25 hours of speech with carefully crafted manual transcriptions, collected from 397 speakers acros"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18584","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.18584/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.18584","created_at":"2026-07-05T10:34:39.780807+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.18584v3","created_at":"2026-07-05T10:34:39.780807+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18584","created_at":"2026-07-05T10:34:39.780807+00:00"},{"alias_kind":"pith_short_12","alias_value":"G34LGQL4CHZA","created_at":"2026-07-05T10:34:39.780807+00:00"},{"alias_kind":"pith_short_16","alias_value":"G34LGQL4CHZANZJ6","created_at":"2026-07-05T10:34:39.780807+00:00"},{"alias_kind":"pith_short_8","alias_value":"G34LGQL4","created_at":"2026-07-05T10:34:39.780807+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.00688","citing_title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19949","citing_title":"Indic-CodecFake meets SATYAM: Towards Detecting Neural Audio Codec Synthesized Speech Deepfakes in Indic Languages","ref_index":117,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX","json":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX.json","graph_json":"https://pith.science/api/pith-number/G34LGQL4CHZANZJ6MZ6OFGDKHX/graph.json","events_json":"https://pith.science/api/pith-number/G34LGQL4CHZANZJ6MZ6OFGDKHX/events.json","paper":"https://pith.science/paper/G34LGQL4"},"agent_actions":{"view_html":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX","download_json":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX.json","view_paper":"https://pith.science/paper/G34LGQL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.18584&json=true","fetch_graph":"https://pith.science/api/pith-number/G34LGQL4CHZANZJ6MZ6OFGDKHX/graph.json","fetch_events":"https://pith.science/api/pith-number/G34LGQL4CHZANZJ6MZ6OFGDKHX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX/action/storage_attestation","attest_author":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX/action/author_attestation","sign_citation":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX/action/citation_signature","submit_replication":"https://pith.science/pith/G34LGQL4CHZANZJ6MZ6OFGDKHX/action/replication_record"}},"created_at":"2026-07-05T10:34:39.780807+00:00","updated_at":"2026-07-05T10:34:39.780807+00:00"}