{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FG3GJ4S7FIGHJ6XOBGQUPVVML4","short_pith_number":"pith:FG3GJ4S7","schema_version":"1.0","canonical_sha256":"29b664f25f2a0c74faee09a147d6ac5f35cdd9f657e8a8fe00daad675a62d067","source":{"kind":"arxiv","id":"2402.01912","version":1},"attestation_state":"computed","paper":{"title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Dan Lyth, Simon King","submitted_at":"2024-02-02T21:29:34Z","abstract_excerpt":"Text-to-speech models trained on large-scale datasets have demonstrated impressive in-context learning capabilities and naturalness. However, control of speaker identity and style in these models typically requires conditioning on reference speech recordings, limiting creative applications. Alternatively, natural language prompting of speaker identity and style has demonstrated promising results and provides an intuitive method of control. However, reliance on human-labeled descriptions prevents scaling to large datasets. Our work bridges the gap between these two approaches. We propose a scal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01912","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2024-02-02T21:29:34Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"3d65d28a05382e9b3c7a3dd2e8f85b068da2bcbdb7c843913ab9855d7312f895","abstract_canon_sha256":"296f74ae1b4a3df09f94741d368b08aa31bd76f809eeedf26e8287abd2437de9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:42:24.176268Z","signature_b64":"AzWD0RcgGPnEeK4C/UbjHWQns0PkG5qZurVpSSQbq29H6R9LdUmnvWr35u1/QhL+XyyTJo/JGVA9Io+76x/rBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"29b664f25f2a0c74faee09a147d6ac5f35cdd9f657e8a8fe00daad675a62d067","last_reissued_at":"2026-07-05T07:42:24.175820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:42:24.175820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Dan Lyth, Simon King","submitted_at":"2024-02-02T21:29:34Z","abstract_excerpt":"Text-to-speech models trained on large-scale datasets have demonstrated impressive in-context learning capabilities and naturalness. However, control of speaker identity and style in these models typically requires conditioning on reference speech recordings, limiting creative applications. Alternatively, natural language prompting of speaker identity and style has demonstrated promising results and provides an intuitive method of control. However, reliance on human-labeled descriptions prevents scaling to large datasets. Our work bridges the gap between these two approaches. We propose a scal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01912","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01912","created_at":"2026-07-05T07:42:24.175876+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01912v1","created_at":"2026-07-05T07:42:24.175876+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01912","created_at":"2026-07-05T07:42:24.175876+00:00"},{"alias_kind":"pith_short_12","alias_value":"FG3GJ4S7FIGH","created_at":"2026-07-05T07:42:24.175876+00:00"},{"alias_kind":"pith_short_16","alias_value":"FG3GJ4S7FIGHJ6XO","created_at":"2026-07-05T07:42:24.175876+00:00"},{"alias_kind":"pith_short_8","alias_value":"FG3GJ4S7","created_at":"2026-07-05T07:42:24.175876+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19209","citing_title":"FineCombo-TTS: Collaborative and Precise Controllable Speech Synthesis Using Text Descriptions and Reference Speech","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12199","citing_title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20650","citing_title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06928","citing_title":"VoxCPM2 Technical Report","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05889","citing_title":"GLASS: GRPO-Trained LoRA for Acoustic Style Steering in Zero-Shot Text-to-Speech","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15621","citing_title":"Qwen3-TTS Technical Report","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02364","citing_title":"When Spoof Detectors Travel: Evaluation Across 66 Languages in the Low-Resource Language Spoofing Corpus","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10117","citing_title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12242","citing_title":"Mind the Pause: Disfluency-Aware Objective Tuning for Multilingual Speech Correction with LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17958","citing_title":"MINT-Bench: A Comprehensive Multilingual Benchmark for Instruction-Following Text-to-Speech","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00861","citing_title":"Voice Mapping of Text-to-Speech Systems: A Metric-Based Approach for Voice Quality Assessment","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19330","citing_title":"Text-To-Speech with Chain-of-Details: modeling temporal dynamics in speech generation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4","json":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4.json","graph_json":"https://pith.science/api/pith-number/FG3GJ4S7FIGHJ6XOBGQUPVVML4/graph.json","events_json":"https://pith.science/api/pith-number/FG3GJ4S7FIGHJ6XOBGQUPVVML4/events.json","paper":"https://pith.science/paper/FG3GJ4S7"},"agent_actions":{"view_html":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4","download_json":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4.json","view_paper":"https://pith.science/paper/FG3GJ4S7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01912&json=true","fetch_graph":"https://pith.science/api/pith-number/FG3GJ4S7FIGHJ6XOBGQUPVVML4/graph.json","fetch_events":"https://pith.science/api/pith-number/FG3GJ4S7FIGHJ6XOBGQUPVVML4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4/action/storage_attestation","attest_author":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4/action/author_attestation","sign_citation":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4/action/citation_signature","submit_replication":"https://pith.science/pith/FG3GJ4S7FIGHJ6XOBGQUPVVML4/action/replication_record"}},"created_at":"2026-07-05T07:42:24.175876+00:00","updated_at":"2026-07-05T07:42:24.175876+00:00"}