{"work":{"id":"6e88ee95-1133-4302-a142-cdf8f9456a8d","openalex_id":"https://openalex.org/W4399425629","doi":"10.48550/arxiv.2406.02430","arxiv_id":"2406.02430","raw_key":null,"title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","authors":null,"authors_text":"Philip Anastassiou, Jiawei Chen, Jitong Chen, Yuanzhe Chen, Zhuo Chen, Ziyi Chen","year":2024,"venue":"eess.AS","abstract":"We introduce Seed-TTS, a family of large-scale autoregressive text-to-speech (TTS) models capable of generating speech that is virtually indistinguishable from human speech. Seed-TTS serves as a foundation model for speech generation and excels in speech in-context learning, achieving performance in speaker similarity and naturalness that matches ground truth human speech in both objective and subjective evaluations. With fine-tuning, we achieve even higher subjective scores across these metrics. Seed-TTS offers superior controllability over various speech attributes such as emotion and is capable of generating highly expressive and diverse speech for speakers in the wild. Furthermore, we propose a self-distillation method for speech factorization, as well as a reinforcement learning approach to enhance model robustness, speaker similarity, and controllability. We additionally present a non-autoregressive (NAR) variant of the Seed-TTS model, named $\\text{Seed-TTS}_\\text{DiT}$, which utilizes a fully diffusion-based architecture. Unlike previous NAR-based TTS systems, $\\text{Seed-TTS}_\\text{DiT}$ does not depend on pre-estimated phoneme durations and performs speech generation through end-to-end processing. We demonstrate that this variant achieves comparable performance to the language model-based variant and showcase its effectiveness in speech editing. We encourage readers to listen to demos at \\url{https://bytedancespeech.github.io/seedtts_tech_report}.","external_url":"https://arxiv.org/abs/2406.02430","cited_by_count":5,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2406.02430","created_at":"2026-05-10T01:04:50.146990+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","render_title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models"},"hub":{"state":{"work_id":"6e88ee95-1133-4302-a142-cdf8f9456a8d","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":77,"external_cited_by_count":5,"distinct_field_count":8,"first_pith_cited_at":"2024-09-27T07:46:52+00:00","last_pith_cited_at":"2026-07-09T09:01:03+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-23T04:19:28.398203+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":4},{"context_role":"dataset","n":4},{"context_role":"method","n":2},{"context_role":"baseline","n":1},{"context_role":"extension","n":1}],"polarity_counts":[{"context_polarity":"use_dataset","n":4},{"context_polarity":"background","n":3},{"context_polarity":"use_method","n":2},{"context_polarity":"baseline","n":1},{"context_polarity":"extend","n":1},{"context_polarity":"unclear","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}