{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:DS4NTRONIXVO3RGPY42DZ5LKM3","short_pith_number":"pith:DS4NTRON","schema_version":"1.0","canonical_sha256":"1cb8d9c5cd45eaedc4cfc7343cf56a66e857999c6804796b38df7057e8ebfde4","source":{"kind":"arxiv","id":"2204.00291","version":1},"attestation_state":"computed","paper":{"title":"Text-To-Speech Data Augmentation for Low Resource Speech Recognition","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Rodolfo Zevallos","submitted_at":"2022-04-01T08:53:44Z","abstract_excerpt":"Nowadays, the main problem of deep learning techniques used in the development of automatic speech recognition (ASR) models is the lack of transcribed data. The goal of this research is to propose a new data augmentation method to improve ASR models for agglutinative and low-resource languages. This novel data augmentation method generates both synthetic text and synthetic audio. Some experiments were conducted using the corpus of the Quechua language, which is an agglutinative and low-resource language. In this study, a sequence-to-sequence (seq2seq) model was applied to generate synthetic te"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.00291","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2022-04-01T08:53:44Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"a3549ed10e80a91f7e85e9ad36432be8a27b28df0355ac0a5e18b7da33495fe1","abstract_canon_sha256":"994354378f58e096ba25f609254c878386b84da7c2517529cb30d59f4343867c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:45.383739Z","signature_b64":"U2WCLYvMUSCh6W484PCRRWjcOSrYQ2Jodt+A54l6CiZgCQTQKtoQBVk/JI/E/bXOX0g6TTiVMu8Zf/ULZFtOBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1cb8d9c5cd45eaedc4cfc7343cf56a66e857999c6804796b38df7057e8ebfde4","last_reissued_at":"2026-07-05T04:10:45.383347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:45.383347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Text-To-Speech Data Augmentation for Low Resource Speech Recognition","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Rodolfo Zevallos","submitted_at":"2022-04-01T08:53:44Z","abstract_excerpt":"Nowadays, the main problem of deep learning techniques used in the development of automatic speech recognition (ASR) models is the lack of transcribed data. The goal of this research is to propose a new data augmentation method to improve ASR models for agglutinative and low-resource languages. This novel data augmentation method generates both synthetic text and synthetic audio. Some experiments were conducted using the corpus of the Quechua language, which is an agglutinative and low-resource language. In this study, a sequence-to-sequence (seq2seq) model was applied to generate synthetic te"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.00291","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.00291/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.00291","created_at":"2026-07-05T04:10:45.383407+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.00291v1","created_at":"2026-07-05T04:10:45.383407+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.00291","created_at":"2026-07-05T04:10:45.383407+00:00"},{"alias_kind":"pith_short_12","alias_value":"DS4NTRONIXVO","created_at":"2026-07-05T04:10:45.383407+00:00"},{"alias_kind":"pith_short_16","alias_value":"DS4NTRONIXVO3RGP","created_at":"2026-07-05T04:10:45.383407+00:00"},{"alias_kind":"pith_short_8","alias_value":"DS4NTRON","created_at":"2026-07-05T04:10:45.383407+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08208","citing_title":"Diarization-Guided Qwen-ASR Adaptation for Multilingual Two-Speaker Conversational Speech","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19381","citing_title":"Improving Code-Switching ASR with Code-Mixing Guided Synthetic Speech","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2309.12802","citing_title":"Deepfake audio as a data augmentation technique for training automatic speech to text transcription models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3","json":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3.json","graph_json":"https://pith.science/api/pith-number/DS4NTRONIXVO3RGPY42DZ5LKM3/graph.json","events_json":"https://pith.science/api/pith-number/DS4NTRONIXVO3RGPY42DZ5LKM3/events.json","paper":"https://pith.science/paper/DS4NTRON"},"agent_actions":{"view_html":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3","download_json":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3.json","view_paper":"https://pith.science/paper/DS4NTRON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.00291&json=true","fetch_graph":"https://pith.science/api/pith-number/DS4NTRONIXVO3RGPY42DZ5LKM3/graph.json","fetch_events":"https://pith.science/api/pith-number/DS4NTRONIXVO3RGPY42DZ5LKM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3/action/storage_attestation","attest_author":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3/action/author_attestation","sign_citation":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3/action/citation_signature","submit_replication":"https://pith.science/pith/DS4NTRONIXVO3RGPY42DZ5LKM3/action/replication_record"}},"created_at":"2026-07-05T04:10:45.383407+00:00","updated_at":"2026-07-05T04:10:45.383407+00:00"}