{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:6RGVSEFEO6PEN3R7JTQTAW3GT2","short_pith_number":"pith:6RGVSEFE","schema_version":"1.0","canonical_sha256":"f44d5910a4779e46ee3f4ce1305b669e945872cfd6f0c7696d733f0209659138","source":{"kind":"arxiv","id":"2206.14053","version":2},"attestation_state":"computed","paper":{"title":"Bengali Common Voice Speech Dataset for Automatic Speech Recognition","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Ahmed Imtiaz Humayun, Asif Sushmit, MD. Nazmuddoha Ansary, Samiul Alam, Sazia Morshed Mehnaz, Shahrin Nakkhatra, Syed Mobassir Hossen, Tahsin Reasat, Zaowad Abdullah","submitted_at":"2022-06-28T14:52:08Z","abstract_excerpt":"Bengali is one of the most spoken languages in the world with over 300 million speakers globally. Despite its popularity, research into the development of Bengali speech recognition systems is hindered due to the lack of diverse open-source datasets. As a way forward, we have crowdsourced the Bengali Common Voice Speech Dataset, which is a sentence-level automatic speech recognition corpus. Collected on the Mozilla Common Voice platform, the dataset is part of an ongoing campaign that has led to the collection of over 400 hours of data in 2 months and is growing rapidly. Our analysis shows tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.14053","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-06-28T14:52:08Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"c2bc108f3b42ba1e75783f8393f618a1561b0c35ca891f75b594707580523097","abstract_canon_sha256":"12793433e6236356cf4eac8a1fdcbed9ceae36833a8669a674137670c038753e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:36:01.935019Z","signature_b64":"JtsoJiOz1J4cd0A2GaBj56mnbzuk+CCsGNXVPxaBscRGtaaecJyApbdLJ9amyIHdAW0qvznjN6TwfEc5u2FDCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f44d5910a4779e46ee3f4ce1305b669e945872cfd6f0c7696d733f0209659138","last_reissued_at":"2026-07-05T04:36:01.934556Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:36:01.934556Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bengali Common Voice Speech Dataset for Automatic Speech Recognition","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Ahmed Imtiaz Humayun, Asif Sushmit, MD. Nazmuddoha Ansary, Samiul Alam, Sazia Morshed Mehnaz, Shahrin Nakkhatra, Syed Mobassir Hossen, Tahsin Reasat, Zaowad Abdullah","submitted_at":"2022-06-28T14:52:08Z","abstract_excerpt":"Bengali is one of the most spoken languages in the world with over 300 million speakers globally. Despite its popularity, research into the development of Bengali speech recognition systems is hindered due to the lack of diverse open-source datasets. As a way forward, we have crowdsourced the Bengali Common Voice Speech Dataset, which is a sentence-level automatic speech recognition corpus. Collected on the Mozilla Common Voice platform, the dataset is part of an ongoing campaign that has led to the collection of over 400 hours of data in 2 months and is growing rapidly. Our analysis shows tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.14053","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.14053/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.14053","created_at":"2026-07-05T04:36:01.934613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.14053v2","created_at":"2026-07-05T04:36:01.934613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.14053","created_at":"2026-07-05T04:36:01.934613+00:00"},{"alias_kind":"pith_short_12","alias_value":"6RGVSEFEO6PE","created_at":"2026-07-05T04:36:01.934613+00:00"},{"alias_kind":"pith_short_16","alias_value":"6RGVSEFEO6PEN3R7","created_at":"2026-07-05T04:36:01.934613+00:00"},{"alias_kind":"pith_short_8","alias_value":"6RGVSEFE","created_at":"2026-07-05T04:36:01.934613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.00733","citing_title":"Quantifying and Reducing Speaker Heterogeneity within the Common Voice Corpus for Phonetic Analysis","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2","json":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2.json","graph_json":"https://pith.science/api/pith-number/6RGVSEFEO6PEN3R7JTQTAW3GT2/graph.json","events_json":"https://pith.science/api/pith-number/6RGVSEFEO6PEN3R7JTQTAW3GT2/events.json","paper":"https://pith.science/paper/6RGVSEFE"},"agent_actions":{"view_html":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2","download_json":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2.json","view_paper":"https://pith.science/paper/6RGVSEFE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.14053&json=true","fetch_graph":"https://pith.science/api/pith-number/6RGVSEFEO6PEN3R7JTQTAW3GT2/graph.json","fetch_events":"https://pith.science/api/pith-number/6RGVSEFEO6PEN3R7JTQTAW3GT2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2/action/storage_attestation","attest_author":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2/action/author_attestation","sign_citation":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2/action/citation_signature","submit_replication":"https://pith.science/pith/6RGVSEFEO6PEN3R7JTQTAW3GT2/action/replication_record"}},"created_at":"2026-07-05T04:36:01.934613+00:00","updated_at":"2026-07-05T04:36:01.934613+00:00"}