{"work":{"id":"de8fb688-dd63-4942-92f6-66d98d5b6db2","openalex_id":"https://openalex.org/W4313679638","doi":"10.48550/arxiv.2301.02111","arxiv_id":"2301.02111","raw_key":null,"title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","authors":null,"authors_text":"Chengyi Wang, Sanyuan Chen, Yu Wu, Ziqiang Zhang, Long Zhou, Shujie Liu","year":2023,"venue":"cs.CL","abstract":"We introduce a language modeling approach for text to speech synthesis (TTS). Specifically, we train a neural codec language model (called Vall-E) using discrete codes derived from an off-the-shelf neural audio codec model, and regard TTS as a conditional language modeling task rather than continuous signal regression as in previous work. During the pre-training stage, we scale up the TTS training data to 60K hours of English speech which is hundreds of times larger than existing systems. Vall-E emerges in-context learning capabilities and can be used to synthesize high-quality personalized speech with only a 3-second enrolled recording of an unseen speaker as an acoustic prompt. Experiment results show that Vall-E significantly outperforms the state-of-the-art zero-shot TTS system in terms of speech naturalness and speaker similarity. In addition, we find Vall-E could preserve the speaker's emotion and acoustic environment of the acoustic prompt in synthesis. See https://aka.ms/valle for demos of our work.","external_url":"https://arxiv.org/abs/2301.02111","cited_by_count":162,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2301.02111","created_at":"2026-05-10T00:34:47.286917+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","render_title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers"},"hub":{"state":{"work_id":"de8fb688-dd63-4942-92f6-66d98d5b6db2","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":95,"external_cited_by_count":162,"distinct_field_count":8,"first_pith_cited_at":"2023-02-27T18:55:27+00:00","last_pith_cited_at":"2026-07-07T09:31:19+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T07:39:26.684550+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":9}],"polarity_counts":[{"context_polarity":"background","n":8},{"context_polarity":"support","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}