{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:7XI46ASNPCCQ3P3HJR66GNB74O","short_pith_number":"pith:7XI46ASN","schema_version":"1.0","canonical_sha256":"fdd1cf024d78850dbf674c7de3343fe3b3e681e8281fa746c14d8b767210ab58","source":{"kind":"arxiv","id":"2010.06778","version":1},"attestation_state":"computed","paper":{"title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alena Butryna, Alexander Gutkin, Anna Katanova, Chenfang Li, Cibu Johny, Clara Rivera, Fei He, Isin Demirsahin, Jaka Aris Eko Wibawa, Keshan Sodimana, Knot Pipatsrisawat, Linne Ha, Martin Jansche, Oddur Kjartansson, Pasindu de Silva, Richard Sproat, Shan-Hui Cathy Chu, Supheakmungkol Sarin, Tatiana Merkulova, Theeraphol Wattanavekin, Yin May Oo","submitted_at":"2020-10-14T02:24:04Z","abstract_excerpt":"This paper presents an overview of a program designed to address the growing need for developing freely available speech resources for under-represented languages. At present we have released 38 datasets for building text-to-speech and automatic speech recognition applications for languages and dialects of South and Southeast Asia, Africa, Europe and South America. The paper describes the methodology used for developing such corpora and presents some of our findings that could benefit under-represented language communities."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.06778","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2020-10-14T02:24:04Z","cross_cats_sorted":[],"title_canon_sha256":"4672a0cd2bc6fe93a410bff84770b7084cb2ce4aabf863b0cea6258ce07b7f4b","abstract_canon_sha256":"e77b57e1198b55121b4297a3558064aa6e97939e13502a6f3b1279002e298b02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:42:59.244808Z","signature_b64":"f1oWOw8DeirIun6WiXElclcMQG3jetybGZxhWcXCCdupDm8IkHHxfjqbqgWPr0N0RuakgDeVX0Z+QwEwa4oMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdd1cf024d78850dbf674c7de3343fe3b3e681e8281fa746c14d8b767210ab58","last_reissued_at":"2026-07-05T01:42:59.244310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:42:59.244310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alena Butryna, Alexander Gutkin, Anna Katanova, Chenfang Li, Cibu Johny, Clara Rivera, Fei He, Isin Demirsahin, Jaka Aris Eko Wibawa, Keshan Sodimana, Knot Pipatsrisawat, Linne Ha, Martin Jansche, Oddur Kjartansson, Pasindu de Silva, Richard Sproat, Shan-Hui Cathy Chu, Supheakmungkol Sarin, Tatiana Merkulova, Theeraphol Wattanavekin, Yin May Oo","submitted_at":"2020-10-14T02:24:04Z","abstract_excerpt":"This paper presents an overview of a program designed to address the growing need for developing freely available speech resources for under-represented languages. At present we have released 38 datasets for building text-to-speech and automatic speech recognition applications for languages and dialects of South and Southeast Asia, Africa, Europe and South America. The paper describes the methodology used for developing such corpora and presents some of our findings that could benefit under-represented language communities."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.06778","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.06778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.06778","created_at":"2026-07-05T01:42:59.244383+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.06778v1","created_at":"2026-07-05T01:42:59.244383+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.06778","created_at":"2026-07-05T01:42:59.244383+00:00"},{"alias_kind":"pith_short_12","alias_value":"7XI46ASNPCCQ","created_at":"2026-07-05T01:42:59.244383+00:00"},{"alias_kind":"pith_short_16","alias_value":"7XI46ASNPCCQ3P3H","created_at":"2026-07-05T01:42:59.244383+00:00"},{"alias_kind":"pith_short_8","alias_value":"7XI46ASN","created_at":"2026-07-05T01:42:59.244383+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O","json":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O.json","graph_json":"https://pith.science/api/pith-number/7XI46ASNPCCQ3P3HJR66GNB74O/graph.json","events_json":"https://pith.science/api/pith-number/7XI46ASNPCCQ3P3HJR66GNB74O/events.json","paper":"https://pith.science/paper/7XI46ASN"},"agent_actions":{"view_html":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O","download_json":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O.json","view_paper":"https://pith.science/paper/7XI46ASN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.06778&json=true","fetch_graph":"https://pith.science/api/pith-number/7XI46ASNPCCQ3P3HJR66GNB74O/graph.json","fetch_events":"https://pith.science/api/pith-number/7XI46ASNPCCQ3P3HJR66GNB74O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O/action/storage_attestation","attest_author":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O/action/author_attestation","sign_citation":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O/action/citation_signature","submit_replication":"https://pith.science/pith/7XI46ASNPCCQ3P3HJR66GNB74O/action/replication_record"}},"created_at":"2026-07-05T01:42:59.244383+00:00","updated_at":"2026-07-05T01:42:59.244383+00:00"}