{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HFJQH56SIQ5CU7MM7K3SACAUUG","short_pith_number":"pith:HFJQH56S","schema_version":"1.0","canonical_sha256":"395303f7d2443a2a7d8cfab7200814a187d123e42c2ead567d58d0e8ff4fb254","source":{"kind":"arxiv","id":"2305.13516","version":1},"attestation_state":"computed","paper":{"title":"Scaling Speech Technology to 1,000+ Languages","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Alexei Baevski, Alexis Conneau, Ali Elkahky, Andros Tjandra, Apoorv Vyas, Arun Babu, Bowen Shi, Maryam Fazel-Zarandi, Michael Auli, Paden Tomasello, Sayani Kundu, Vineel Pratap, Wei-Ning Hsu, Xiaohui Zhang, Yossi Adi, Zhaoheng Ni","submitted_at":"2023-05-22T22:09:41Z","abstract_excerpt":"Expanding the language coverage of speech technology has the potential to improve access to information for many more people. However, current speech technology is restricted to about one hundred languages which is a small fraction of the over 7,000 languages spoken around the world. The Massively Multilingual Speech (MMS) project increases the number of supported languages by 10-40x, depending on the task. The main ingredients are a new dataset based on readings of publicly available religious texts and effectively leveraging self-supervised learning. We built pre-trained wav2vec 2.0 models c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13516","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-22T22:09:41Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"c34b529166e0f79bad2d1cb9ebb7a66c4dc9827dcd1b33fec115816d3b981272","abstract_canon_sha256":"37bdc890433ae4fa0dc4e584b504b27219404a2514ea6b41ed0740f0adaa36e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:53.844607Z","signature_b64":"Gc3JSygRRxqBEB0P/SoM5FUqhWHPQs0Tf3ychx0iFOnHtuPd3443SZpUCafFQ7ys0iIzmuj8R2c998J5IsvdCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"395303f7d2443a2a7d8cfab7200814a187d123e42c2ead567d58d0e8ff4fb254","last_reissued_at":"2026-07-05T06:12:53.844159Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:53.844159Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Speech Technology to 1,000+ Languages","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Alexei Baevski, Alexis Conneau, Ali Elkahky, Andros Tjandra, Apoorv Vyas, Arun Babu, Bowen Shi, Maryam Fazel-Zarandi, Michael Auli, Paden Tomasello, Sayani Kundu, Vineel Pratap, Wei-Ning Hsu, Xiaohui Zhang, Yossi Adi, Zhaoheng Ni","submitted_at":"2023-05-22T22:09:41Z","abstract_excerpt":"Expanding the language coverage of speech technology has the potential to improve access to information for many more people. However, current speech technology is restricted to about one hundred languages which is a small fraction of the over 7,000 languages spoken around the world. The Massively Multilingual Speech (MMS) project increases the number of supported languages by 10-40x, depending on the task. The main ingredients are a new dataset based on readings of publicly available religious texts and effectively leveraging self-supervised learning. We built pre-trained wav2vec 2.0 models c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13516","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13516/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13516","created_at":"2026-07-05T06:12:53.844218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13516v1","created_at":"2026-07-05T06:12:53.844218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13516","created_at":"2026-07-05T06:12:53.844218+00:00"},{"alias_kind":"pith_short_12","alias_value":"HFJQH56SIQ5C","created_at":"2026-07-05T06:12:53.844218+00:00"},{"alias_kind":"pith_short_16","alias_value":"HFJQH56SIQ5CU7MM","created_at":"2026-07-05T06:12:53.844218+00:00"},{"alias_kind":"pith_short_8","alias_value":"HFJQH56S","created_at":"2026-07-05T06:12:53.844218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26618","citing_title":"Closing the Quality Gap in Low-Resource Text-to-Speech: LoRA Fine-Tuning of VoxCPM2 for Khmer and Korean","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11542","citing_title":"Pretrained self-supervised speech models can recognize unseen consonants","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09317","citing_title":"A Comparative Study of Pre-trained Speech Encoders and Training Objectives for Large-Scale Indic Spoken Language Identification","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2401.09512","citing_title":"MLAAD: The Multi-Language Audio Anti-Spoofing Dataset","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20481","citing_title":"Coherence in the brain unfolds across separable temporal regimes","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02374","citing_title":"Evaluating Generalization and Robustness in Russian Anti-Spoofing: The RuASD Initiative","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18932","citing_title":"Tadabur: A Large-Scale Quran Audio Dataset","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10736","citing_title":"BlasBench: An Open Benchmark for Irish Speech Recognition","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10123","citing_title":"Training-Free Cross-Lingual Dysarthria Severity Assessment via Phonological Subspace Analysis in Self-Supervised Speech Representations","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04598","citing_title":"Benchmarking Multilingual Speech Models on Pashto: Zero-Shot ASR, Script Failure, and Cross-Domain Evaluation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01381","citing_title":"A framework for analyzing concept representations in neural models","ref_index":189,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG","json":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG.json","graph_json":"https://pith.science/api/pith-number/HFJQH56SIQ5CU7MM7K3SACAUUG/graph.json","events_json":"https://pith.science/api/pith-number/HFJQH56SIQ5CU7MM7K3SACAUUG/events.json","paper":"https://pith.science/paper/HFJQH56S"},"agent_actions":{"view_html":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG","download_json":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG.json","view_paper":"https://pith.science/paper/HFJQH56S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13516&json=true","fetch_graph":"https://pith.science/api/pith-number/HFJQH56SIQ5CU7MM7K3SACAUUG/graph.json","fetch_events":"https://pith.science/api/pith-number/HFJQH56SIQ5CU7MM7K3SACAUUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG/action/storage_attestation","attest_author":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG/action/author_attestation","sign_citation":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG/action/citation_signature","submit_replication":"https://pith.science/pith/HFJQH56SIQ5CU7MM7K3SACAUUG/action/replication_record"}},"created_at":"2026-07-05T06:12:53.844218+00:00","updated_at":"2026-07-05T06:12:53.844218+00:00"}