{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BAX4XAJZXCMSJ5HYT27VKRKL57","short_pith_number":"pith:BAX4XAJZ","schema_version":"1.0","canonical_sha256":"082fcb8139b89924f4f89ebf55454beff81a4935d6ef80a160ca767172ca807a","source":{"kind":"arxiv","id":"2306.00998","version":1},"attestation_state":"computed","paper":{"title":"Towards Selection of Text-to-speech Data to Augment ASR Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chunyang Wu, Gil Keren, Jay Mahadeokar, Leda Sar{\\i}, Ozlem Kalinli, Shuo Liu, Yuan Shangguan","submitted_at":"2023-05-30T17:24:28Z","abstract_excerpt":"This paper presents a method for selecting appropriate synthetic speech samples from a given large text-to-speech (TTS) dataset as supplementary training data for an automatic speech recognition (ASR) model. We trained a neural network, which can be optimised using cross-entropy loss or Arcface loss, to measure the similarity of a synthetic data to real speech. We found that incorporating synthetic samples with considerable dissimilarity to real speech, owing in part to lexical differences, into ASR training is crucial for boosting recognition performance. Experimental results on Librispeech t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.00998","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2023-05-30T17:24:28Z","cross_cats_sorted":["cs.CL","cs.SD"],"title_canon_sha256":"0f99e682f829cc354322ab7328ca36232a722cc20d97c259d518745d69ac3003","abstract_canon_sha256":"656bb1a7910da88ab46065b3c8d1382c621e1c2938e6f614b0e1165978005c2c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:36.840779Z","signature_b64":"KHqvGlKnvpHMXC/BE241q57GKpqHjOoN1SZKljTRN7L7OwRIgDp22b0J9isGcCMz/EVpdVusFtq3TUkGDw1QDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"082fcb8139b89924f4f89ebf55454beff81a4935d6ef80a160ca767172ca807a","last_reissued_at":"2026-07-05T06:16:36.840367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:36.840367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Selection of Text-to-speech Data to Augment ASR Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chunyang Wu, Gil Keren, Jay Mahadeokar, Leda Sar{\\i}, Ozlem Kalinli, Shuo Liu, Yuan Shangguan","submitted_at":"2023-05-30T17:24:28Z","abstract_excerpt":"This paper presents a method for selecting appropriate synthetic speech samples from a given large text-to-speech (TTS) dataset as supplementary training data for an automatic speech recognition (ASR) model. We trained a neural network, which can be optimised using cross-entropy loss or Arcface loss, to measure the similarity of a synthetic data to real speech. We found that incorporating synthetic samples with considerable dissimilarity to real speech, owing in part to lexical differences, into ASR training is crucial for boosting recognition performance. Experimental results on Librispeech t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00998","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00998/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.00998","created_at":"2026-07-05T06:16:36.840428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.00998v1","created_at":"2026-07-05T06:16:36.840428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00998","created_at":"2026-07-05T06:16:36.840428+00:00"},{"alias_kind":"pith_short_12","alias_value":"BAX4XAJZXCMS","created_at":"2026-07-05T06:16:36.840428+00:00"},{"alias_kind":"pith_short_16","alias_value":"BAX4XAJZXCMSJ5HY","created_at":"2026-07-05T06:16:36.840428+00:00"},{"alias_kind":"pith_short_8","alias_value":"BAX4XAJZ","created_at":"2026-07-05T06:16:36.840428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57","json":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57.json","graph_json":"https://pith.science/api/pith-number/BAX4XAJZXCMSJ5HYT27VKRKL57/graph.json","events_json":"https://pith.science/api/pith-number/BAX4XAJZXCMSJ5HYT27VKRKL57/events.json","paper":"https://pith.science/paper/BAX4XAJZ"},"agent_actions":{"view_html":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57","download_json":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57.json","view_paper":"https://pith.science/paper/BAX4XAJZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.00998&json=true","fetch_graph":"https://pith.science/api/pith-number/BAX4XAJZXCMSJ5HYT27VKRKL57/graph.json","fetch_events":"https://pith.science/api/pith-number/BAX4XAJZXCMSJ5HYT27VKRKL57/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57/action/storage_attestation","attest_author":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57/action/author_attestation","sign_citation":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57/action/citation_signature","submit_replication":"https://pith.science/pith/BAX4XAJZXCMSJ5HYT27VKRKL57/action/replication_record"}},"created_at":"2026-07-05T06:16:36.840428+00:00","updated_at":"2026-07-05T06:16:36.840428+00:00"}