{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5LMWFQSH6ORXZPONRLWPWW3CMR","short_pith_number":"pith:5LMWFQSH","schema_version":"1.0","canonical_sha256":"ead962c247f3a37cbdcd8aecfb5b626455d23c5d525602a59c32acde2ed12cb9","source":{"kind":"arxiv","id":"2304.09116","version":3},"attestation_state":"computed","paper":{"title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jiang Bian, Kai Shen, Lei He, Sheng Zhao, Tao Qin, Xu Tan, Yanqing Liu, Yichong Leng, Zeqian Ju","submitted_at":"2023-04-18T16:31:59Z","abstract_excerpt":"Scaling text-to-speech (TTS) to large-scale, multi-speaker, and in-the-wild datasets is important to capture the diversity in human speech such as speaker identities, prosodies, and styles (e.g., singing). Current large TTS systems usually quantize speech into discrete tokens and use language models to generate these tokens one by one, which suffer from unstable prosody, word skipping/repeating issue, and poor voice quality. In this paper, we develop NaturalSpeech 2, a TTS system that leverages a neural audio codec with residual vector quantizers to get the quantized latent vectors and uses a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.09116","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-04-18T16:31:59Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG","cs.SD"],"title_canon_sha256":"b99eb1b167d5aa2a7112cc6b4780ed72e2015b50ff1b5362a699040363fdf0f3","abstract_canon_sha256":"51abb41e7566378df97c7b47486de4d690b2df0178da177a9c239f8ef1afe671"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:28.955346Z","signature_b64":"iPp+V/G9jFx+BOBsJQFaGt8V4qhf9S/Mp6+jHjlXKkYZtG+EdNU89iEfFCiI30HmGChBjutEwtvo+6mqtYO2Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ead962c247f3a37cbdcd8aecfb5b626455d23c5d525602a59c32acde2ed12cb9","last_reissued_at":"2026-07-05T06:15:28.954975Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:28.954975Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Jiang Bian, Kai Shen, Lei He, Sheng Zhao, Tao Qin, Xu Tan, Yanqing Liu, Yichong Leng, Zeqian Ju","submitted_at":"2023-04-18T16:31:59Z","abstract_excerpt":"Scaling text-to-speech (TTS) to large-scale, multi-speaker, and in-the-wild datasets is important to capture the diversity in human speech such as speaker identities, prosodies, and styles (e.g., singing). Current large TTS systems usually quantize speech into discrete tokens and use language models to generate these tokens one by one, which suffer from unstable prosody, word skipping/repeating issue, and poor voice quality. In this paper, we develop NaturalSpeech 2, a TTS system that leverages a neural audio codec with residual vector quantizers to get the quantized latent vectors and uses a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.09116","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.09116/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.09116","created_at":"2026-07-05T06:15:28.955031+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.09116v3","created_at":"2026-07-05T06:15:28.955031+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.09116","created_at":"2026-07-05T06:15:28.955031+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LMWFQSH6ORX","created_at":"2026-07-05T06:15:28.955031+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LMWFQSH6ORXZPON","created_at":"2026-07-05T06:15:28.955031+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LMWFQSH","created_at":"2026-07-05T06:15:28.955031+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23489","citing_title":"MeshFlow: Mesh Generation with Equivariant Flow Matching","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20650","citing_title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05852","citing_title":"UniVoice: A Unified Model for Speech and Singing Voice Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16578","citing_title":"Voice \"Cloning\" is Style Transfer","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2506.23552","citing_title":"JAM-Flow: Joint Audio-Motion Synthesis with Flow Matching","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16964","citing_title":"SemaVoice: Semantic-Aware Continuous Autoregressive Speech Synthesis","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18060","citing_title":"TMD-TTS: A Unified Tibetan Multi-Dialect Text-to-Speech Framework for \\\"U-Tsang, Amdo and Kham Speech Dataset Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15621","citing_title":"Qwen3-TTS Technical Report","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06885","citing_title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","ref_index":139,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24416","citing_title":"Scaling Properties of Continuous Diffusion Spoken Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13720","citing_title":"Movie Gen: A Cast of Media Foundation Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06327","citing_title":"A Novel Automatic Framework for Speaker Drift Detection in Synthesized Speech","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR","json":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR.json","graph_json":"https://pith.science/api/pith-number/5LMWFQSH6ORXZPONRLWPWW3CMR/graph.json","events_json":"https://pith.science/api/pith-number/5LMWFQSH6ORXZPONRLWPWW3CMR/events.json","paper":"https://pith.science/paper/5LMWFQSH"},"agent_actions":{"view_html":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR","download_json":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR.json","view_paper":"https://pith.science/paper/5LMWFQSH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.09116&json=true","fetch_graph":"https://pith.science/api/pith-number/5LMWFQSH6ORXZPONRLWPWW3CMR/graph.json","fetch_events":"https://pith.science/api/pith-number/5LMWFQSH6ORXZPONRLWPWW3CMR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR/action/storage_attestation","attest_author":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR/action/author_attestation","sign_citation":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR/action/citation_signature","submit_replication":"https://pith.science/pith/5LMWFQSH6ORXZPONRLWPWW3CMR/action/replication_record"}},"created_at":"2026-07-05T06:15:28.955031+00:00","updated_at":"2026-07-05T06:15:28.955031+00:00"}