{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PYCFUJ5B44SJUZPMFJXE6BP7MO","short_pith_number":"pith:PYCFUJ5B","schema_version":"1.0","canonical_sha256":"7e045a27a1e7249a65ec2a6e4f05ff639d8bf693610a1774a77214ca451bab2a","source":{"kind":"arxiv","id":"2407.04047","version":1},"attestation_state":"computed","paper":{"title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Cong-Thanh Do, Rama Doddipatla, Shuhei Imai, Thomas Hain","submitted_at":"2024-07-04T16:42:24Z","abstract_excerpt":"This paper investigates the use of unsupervised text-to-speech synthesis (TTS) as a data augmentation method to improve accented speech recognition. TTS systems are trained with a small amount of accented speech training data and their pseudo-labels rather than manual transcriptions, and hence unsupervised. This approach enables the use of accented speech data without manual transcriptions to perform data augmentation for accented speech recognition. Synthetic accented speech data, generated from text prompts by using the TTS systems, are then combined with available non-accented speech data t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04047","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-04T16:42:24Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"f0a3bd5ff69c0ce834895dfca742df490657906a4173c4f7d73525f4ccc149d2","abstract_canon_sha256":"c8cc31dce4b92c9605baa767063c23a3efe1b990e2d758d35a05d3b87e84b2eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:27.920296Z","signature_b64":"qfiLiMF0an1/23lvepax5QqF9Pq6dnTxDQ6mhh9I2S8jAJDV9F10TSdRe0kpkNitN/AGyOQM6K6E5YkEuFYXDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e045a27a1e7249a65ec2a6e4f05ff639d8bf693610a1774a77214ca451bab2a","last_reissued_at":"2026-07-05T08:40:27.919835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:27.919835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Cong-Thanh Do, Rama Doddipatla, Shuhei Imai, Thomas Hain","submitted_at":"2024-07-04T16:42:24Z","abstract_excerpt":"This paper investigates the use of unsupervised text-to-speech synthesis (TTS) as a data augmentation method to improve accented speech recognition. TTS systems are trained with a small amount of accented speech training data and their pseudo-labels rather than manual transcriptions, and hence unsupervised. This approach enables the use of accented speech data without manual transcriptions to perform data augmentation for accented speech recognition. Synthetic accented speech data, generated from text prompts by using the TTS systems, are then combined with available non-accented speech data t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04047","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04047","created_at":"2026-07-05T08:40:27.919890+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04047v1","created_at":"2026-07-05T08:40:27.919890+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04047","created_at":"2026-07-05T08:40:27.919890+00:00"},{"alias_kind":"pith_short_12","alias_value":"PYCFUJ5B44SJ","created_at":"2026-07-05T08:40:27.919890+00:00"},{"alias_kind":"pith_short_16","alias_value":"PYCFUJ5B44SJUZPM","created_at":"2026-07-05T08:40:27.919890+00:00"},{"alias_kind":"pith_short_8","alias_value":"PYCFUJ5B","created_at":"2026-07-05T08:40:27.919890+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20606","citing_title":"Towards Pretraining Robust ASR Foundation Model with Acoustic-Aware Data Augmentation","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO","json":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO.json","graph_json":"https://pith.science/api/pith-number/PYCFUJ5B44SJUZPMFJXE6BP7MO/graph.json","events_json":"https://pith.science/api/pith-number/PYCFUJ5B44SJUZPMFJXE6BP7MO/events.json","paper":"https://pith.science/paper/PYCFUJ5B"},"agent_actions":{"view_html":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO","download_json":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO.json","view_paper":"https://pith.science/paper/PYCFUJ5B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04047&json=true","fetch_graph":"https://pith.science/api/pith-number/PYCFUJ5B44SJUZPMFJXE6BP7MO/graph.json","fetch_events":"https://pith.science/api/pith-number/PYCFUJ5B44SJUZPMFJXE6BP7MO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO/action/storage_attestation","attest_author":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO/action/author_attestation","sign_citation":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO/action/citation_signature","submit_replication":"https://pith.science/pith/PYCFUJ5B44SJUZPMFJXE6BP7MO/action/replication_record"}},"created_at":"2026-07-05T08:40:27.919890+00:00","updated_at":"2026-07-05T08:40:27.919890+00:00"}