{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5YBVB564HDMGGVJEROWQ3TL7BL","short_pith_number":"pith:5YBVB564","schema_version":"1.0","canonical_sha256":"ee0350f7dc38d86355248bad0dcd7f0ad59c95317ef8c21bff9f6be8f92035fb","source":{"kind":"arxiv","id":"2412.06602","version":3},"attestation_state":"computed","paper":{"title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Li Liu, Pengfei Zhang, Tianxin Xie, Wenwu Wang, Yan Rong","submitted_at":"2024-12-09T15:50:25Z","abstract_excerpt":"Text-to-speech (TTS) has advanced from generating natural-sounding speech to enabling fine-grained control over attributes like emotion, timbre, and style. Driven by rising industrial demand and breakthroughs in deep learning, e.g., diffusion and large language models (LLMs), controllable TTS has become a rapidly growing research area. This survey provides the first comprehensive review of controllable TTS methods, from traditional control techniques to emerging approaches using natural language prompts. We categorize model architectures, control strategies, and feature representations, while "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.06602","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-09T15:50:25Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"0bda23c3b911e2395401ce8081eb9cbe6fe545e3d1e5cfd16dc366da0e034319","abstract_canon_sha256":"12a0b75c44d67e794e83a153a14fc3187e8ac03a0b9473257a53972f2c42f173"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:22.609593Z","signature_b64":"/w9WRzjTAz2lD680B4SgSTKG1k7mjBtiNu8ghdryAV5Ne+wMjL/GbjIBL+qVQYoPgNsISSBfV73M54Hk0YeQAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee0350f7dc38d86355248bad0dcd7f0ad59c95317ef8c21bff9f6be8f92035fb","last_reissued_at":"2026-07-05T11:58:22.609077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:22.609077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Controllable Speech Synthesis in the Era of Large Language Models: A Systematic Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Li Liu, Pengfei Zhang, Tianxin Xie, Wenwu Wang, Yan Rong","submitted_at":"2024-12-09T15:50:25Z","abstract_excerpt":"Text-to-speech (TTS) has advanced from generating natural-sounding speech to enabling fine-grained control over attributes like emotion, timbre, and style. Driven by rising industrial demand and breakthroughs in deep learning, e.g., diffusion and large language models (LLMs), controllable TTS has become a rapidly growing research area. This survey provides the first comprehensive review of controllable TTS methods, from traditional control techniques to emerging approaches using natural language prompts. We categorize model architectures, control strategies, and feature representations, while "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.06602","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.06602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.06602","created_at":"2026-07-05T11:58:22.609138+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.06602v3","created_at":"2026-07-05T11:58:22.609138+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.06602","created_at":"2026-07-05T11:58:22.609138+00:00"},{"alias_kind":"pith_short_12","alias_value":"5YBVB564HDMG","created_at":"2026-07-05T11:58:22.609138+00:00"},{"alias_kind":"pith_short_16","alias_value":"5YBVB564HDMGGVJE","created_at":"2026-07-05T11:58:22.609138+00:00"},{"alias_kind":"pith_short_8","alias_value":"5YBVB564","created_at":"2026-07-05T11:58:22.609138+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24618","citing_title":"FC-TTS: Style and Timbre Control in Zero-Shot Text-to-Speech with Disentangled Speech Representations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06201","citing_title":"TokenChain: A Discrete Speech Chain via Semantic Token Modeling","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08363","citing_title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL","json":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL.json","graph_json":"https://pith.science/api/pith-number/5YBVB564HDMGGVJEROWQ3TL7BL/graph.json","events_json":"https://pith.science/api/pith-number/5YBVB564HDMGGVJEROWQ3TL7BL/events.json","paper":"https://pith.science/paper/5YBVB564"},"agent_actions":{"view_html":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL","download_json":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL.json","view_paper":"https://pith.science/paper/5YBVB564","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.06602&json=true","fetch_graph":"https://pith.science/api/pith-number/5YBVB564HDMGGVJEROWQ3TL7BL/graph.json","fetch_events":"https://pith.science/api/pith-number/5YBVB564HDMGGVJEROWQ3TL7BL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL/action/storage_attestation","attest_author":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL/action/author_attestation","sign_citation":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL/action/citation_signature","submit_replication":"https://pith.science/pith/5YBVB564HDMGGVJEROWQ3TL7BL/action/replication_record"}},"created_at":"2026-07-05T11:58:22.609138+00:00","updated_at":"2026-07-05T11:58:22.609138+00:00"}