{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KYN77HD5I7GCCUFI4ZDCFHKSVO","short_pith_number":"pith:KYN77HD5","schema_version":"1.0","canonical_sha256":"561bff9c7d47cc2150a8e646229d52ab96a7fa2f0d81f8e7e7d2d40053af0c08","source":{"kind":"arxiv","id":"2406.05370","version":2},"attestation_state":"computed","paper":{"title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Jinyu Li, Long Zhou, Sanyuan Chen, Sheng Zhao, Shujie Liu, Xu Tan, Yanqing Liu, Yao Qian","submitted_at":"2024-06-08T06:31:03Z","abstract_excerpt":"This paper introduces VALL-E 2, the latest advancement in neural codec language models that marks a milestone in zero-shot text-to-speech synthesis (TTS), achieving human parity for the first time. Based on its predecessor, VALL-E, the new iteration introduces two significant enhancements: Repetition Aware Sampling refines the original nucleus sampling process by accounting for token repetition in the decoding history. It not only stabilizes the decoding but also circumvents the infinite loop issue. Grouped Code Modeling organizes codec codes into groups to effectively shorten the sequence len"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05370","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-08T06:31:03Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"5c10f188e87563a3e6466f949a077983a495929600773485f6861524f7f096ec","abstract_canon_sha256":"f44e08f55e704d9a5b4a47025e8f7b5786372de36925eeb9eac259751fe30988"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:41.932715Z","signature_b64":"Vry+6p9+keI2GrPf8D8tLwt6TOXAj1ZDlJVtWE5l32ihDTpxqtBbCVQJaA/UeO7kxTL4pMGnb0Xg4TgfHCP7BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"561bff9c7d47cc2150a8e646229d52ab96a7fa2f0d81f8e7e7d2d40053af0c08","last_reissued_at":"2026-07-05T08:32:41.932187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:41.932187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Jinyu Li, Long Zhou, Sanyuan Chen, Sheng Zhao, Shujie Liu, Xu Tan, Yanqing Liu, Yao Qian","submitted_at":"2024-06-08T06:31:03Z","abstract_excerpt":"This paper introduces VALL-E 2, the latest advancement in neural codec language models that marks a milestone in zero-shot text-to-speech synthesis (TTS), achieving human parity for the first time. Based on its predecessor, VALL-E, the new iteration introduces two significant enhancements: Repetition Aware Sampling refines the original nucleus sampling process by accounting for token repetition in the decoding history. It not only stabilizes the decoding but also circumvents the infinite loop issue. Grouped Code Modeling organizes codec codes into groups to effectively shorten the sequence len"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05370","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05370/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05370","created_at":"2026-07-05T08:32:41.932269+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05370v2","created_at":"2026-07-05T08:32:41.932269+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05370","created_at":"2026-07-05T08:32:41.932269+00:00"},{"alias_kind":"pith_short_12","alias_value":"KYN77HD5I7GC","created_at":"2026-07-05T08:32:41.932269+00:00"},{"alias_kind":"pith_short_16","alias_value":"KYN77HD5I7GCCUFI","created_at":"2026-07-05T08:32:41.932269+00:00"},{"alias_kind":"pith_short_8","alias_value":"KYN77HD5","created_at":"2026-07-05T08:32:41.932269+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":236,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25403","citing_title":"CrossAccent-TTS: Cross-Lingual Accent-Intensity Controllable Text-to-Speech via Disentangled Speaker and Accent Representations","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20266","citing_title":"Transcript-Free Flow-Matching Text-to-Speech via Speech Feature Conditioning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20137","citing_title":"PASQA: Pitch-Accent-Focused Speech Quality Assessment Model Trained on Synthetic Speech with Accent Errors","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19823","citing_title":"Low-Burden Data Augmentation for Dysarthric ASR via Zero-Shot Voice Cloning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18485","citing_title":"MagpieTTS-LF: Inference-Time Long-Form Speech Generation Without Training on Long-Form data","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09019","citing_title":"TLDR: Compressing Audio Tokens for Efficient Autoregressive Text-to-Speech","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05852","citing_title":"UniVoice: A Unified Model for Speech and Singing Voice Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00387","citing_title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03455","citing_title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01677","citing_title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26136","citing_title":"Eroding Trust in Real Speech: A Large-Scale Study of Human Audio Deepfake Perception","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25669","citing_title":"Ultra-Low-Bitrate Mel-Spectrogram-based Neural Speech Coding with Flow-Matching-based Refinement and Vocoding-driven Reconstruction","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26672","citing_title":"Can We Hear from Events? Generating Speech from Event Camera","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29859","citing_title":"MELD: Mel-Spectrogram-Based Speech Language Modeling with Discrete Latent Variables","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19883","citing_title":"CoMelSinger: Discrete Token-Based Zero-Shot Singing Synthesis With Structured Melody Control and Guidance","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06885","citing_title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17589","citing_title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05373","citing_title":"Hierarchical Decoding for Discrete Speech Synthesis with Multi-Resolution Spoof Detection","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10117","citing_title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26296","citing_title":"SPG-Codec: Exploring the Role and Boundaries of Semantic Priors in Ultra-Low-Bitrate Neural Speech Coding","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11103","citing_title":"ActorMind: Emulating Human Actor Reasoning for Speech Role-Playing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10065","citing_title":"ASPIRin: Action Space Projection for Interactivity-Optimized Reinforcement Learning in Full-Duplex Speech Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08363","citing_title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16056","citing_title":"AST: Adaptive, Seamless, and Training-Free Precise Speech Editing","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO","json":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO.json","graph_json":"https://pith.science/api/pith-number/KYN77HD5I7GCCUFI4ZDCFHKSVO/graph.json","events_json":"https://pith.science/api/pith-number/KYN77HD5I7GCCUFI4ZDCFHKSVO/events.json","paper":"https://pith.science/paper/KYN77HD5"},"agent_actions":{"view_html":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO","download_json":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO.json","view_paper":"https://pith.science/paper/KYN77HD5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05370&json=true","fetch_graph":"https://pith.science/api/pith-number/KYN77HD5I7GCCUFI4ZDCFHKSVO/graph.json","fetch_events":"https://pith.science/api/pith-number/KYN77HD5I7GCCUFI4ZDCFHKSVO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO/action/storage_attestation","attest_author":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO/action/author_attestation","sign_citation":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO/action/citation_signature","submit_replication":"https://pith.science/pith/KYN77HD5I7GCCUFI4ZDCFHKSVO/action/replication_record"}},"created_at":"2026-07-05T08:32:41.932269+00:00","updated_at":"2026-07-05T08:32:41.932269+00:00"}