{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6PICN2B2K624T4CSZSV62VNZJO","short_pith_number":"pith:6PICN2B2","schema_version":"1.0","canonical_sha256":"f3d026e83a57b5c9f052ccabed55b94b98ab8fd98758ec9452c3c2b75194f097","source":{"kind":"arxiv","id":"2409.00750","version":3},"attestation_state":"computed","paper":{"title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haotian Guo, Haoyue Zhan, Jiachen Zheng, Liwei Liu, Qiang Zhang, Ruihong Zeng, Shunsi Zhang, Xueyao Zhang, Yuancheng Wang, Zhizheng Wu","submitted_at":"2024-09-01T15:26:30Z","abstract_excerpt":"The recent large-scale text-to-speech (TTS) systems are usually grouped as autoregressive and non-autoregressive systems. The autoregressive systems implicitly model duration but exhibit certain deficiencies in robustness and lack of duration controllability. Non-autoregressive systems require explicit alignment information between text and speech during training and predict durations for linguistic units (e.g. phone), which may compromise their naturalness. In this paper, we introduce Masked Generative Codec Transformer (MaskGCT), a fully non-autoregressive TTS model that eliminates the need "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00750","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2024-09-01T15:26:30Z","cross_cats_sorted":["cs.AI","cs.LG","eess.AS"],"title_canon_sha256":"977b09d3cd658d86089bb1000d290b488b84e140790147402abc9001dc67f8bc","abstract_canon_sha256":"8accb03e9082b703ba4a72bad7f28aa28187e34c8b3f79b212430405ae738d3b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:11.214176Z","signature_b64":"k6kl/sXUmxLR51cPUNy9lTiaPHkRUmBcdxncbFpY973HLSB+cN4nZoLxhi6T9xCZ0mH3/fW/TJWsD+Bj3wlWBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3d026e83a57b5c9f052ccabed55b94b98ab8fd98758ec9452c3c2b75194f097","last_reissued_at":"2026-07-05T09:23:11.213690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:11.213690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Haotian Guo, Haoyue Zhan, Jiachen Zheng, Liwei Liu, Qiang Zhang, Ruihong Zeng, Shunsi Zhang, Xueyao Zhang, Yuancheng Wang, Zhizheng Wu","submitted_at":"2024-09-01T15:26:30Z","abstract_excerpt":"The recent large-scale text-to-speech (TTS) systems are usually grouped as autoregressive and non-autoregressive systems. The autoregressive systems implicitly model duration but exhibit certain deficiencies in robustness and lack of duration controllability. Non-autoregressive systems require explicit alignment information between text and speech during training and predict durations for linguistic units (e.g. phone), which may compromise their naturalness. In this paper, we introduce Masked Generative Codec Transformer (MaskGCT), a fully non-autoregressive TTS model that eliminates the need "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00750","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00750/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00750","created_at":"2026-07-05T09:23:11.213745+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00750v3","created_at":"2026-07-05T09:23:11.213745+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00750","created_at":"2026-07-05T09:23:11.213745+00:00"},{"alias_kind":"pith_short_12","alias_value":"6PICN2B2K624","created_at":"2026-07-05T09:23:11.213745+00:00"},{"alias_kind":"pith_short_16","alias_value":"6PICN2B2K624T4CS","created_at":"2026-07-05T09:23:11.213745+00:00"},{"alias_kind":"pith_short_8","alias_value":"6PICN2B2","created_at":"2026-07-05T09:23:11.213745+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23190","citing_title":"FlowTTS-GRPO: Online Reinforcement Learning with Multi-Objective Reward Optimization for Flow-Matching Based Text-to-Speech","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21343","citing_title":"An Evaluation Framework for Text-to-Speech Voice Reconstruction","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09141","citing_title":"FlashTTS: Fast Streaming TTS with MTP Acceleration and X-pred Mean Flow Distillation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09234","citing_title":"End-to-End Training for Discrete Token LLM based TTS System","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09019","citing_title":"TLDR: Compressing Audio Tokens for Efficient Autoregressive Text-to-Speech","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00387","citing_title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03455","citing_title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02638","citing_title":"SegTune: Structured and Fine-Grained Control for Song Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30993","citing_title":"SwanVoice: Expressive Long-Form Zero-Shot Speech Synthesis for Both Monologue and Dialogue","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31247","citing_title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","ref_index":202,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17085","citing_title":"Taming Audio VAEs via Target-KL Regularization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09318","citing_title":"ZipVoice-Dialog: Non-Autoregressive Spoken Dialogue Generation with Flow Matching","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06201","citing_title":"TokenChain: A Discrete Speech Chain via Semantic Token Modeling","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15621","citing_title":"Qwen3-TTS Technical Report","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06885","citing_title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2507.16632","citing_title":"Step-Audio 2 Technical Report","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17589","citing_title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10117","citing_title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11098","citing_title":"AffectCodec: Emotion-Preserving Neural Speech Codec for Expressive Speech Modeling","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22209","citing_title":"UniSonate: A Unified Model for Speech, Music, and Sound Effect Generation with Text Instructions","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11552","citing_title":"MimicLM: Zero-Shot Voice Imitation through Autoregressive Modeling of Pseudo-Parallel Speech Corpora","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11283","citing_title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05526","citing_title":"Controllable Singing Style Conversion with Boundary-Aware Information Bottleneck","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15923","citing_title":"Hierarchical Codec Diffusion for Video-to-Speech Generation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19330","citing_title":"Text-To-Speech with Chain-of-Details: modeling temporal dynamics in speech generation","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO","json":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO.json","graph_json":"https://pith.science/api/pith-number/6PICN2B2K624T4CSZSV62VNZJO/graph.json","events_json":"https://pith.science/api/pith-number/6PICN2B2K624T4CSZSV62VNZJO/events.json","paper":"https://pith.science/paper/6PICN2B2"},"agent_actions":{"view_html":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO","download_json":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO.json","view_paper":"https://pith.science/paper/6PICN2B2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00750&json=true","fetch_graph":"https://pith.science/api/pith-number/6PICN2B2K624T4CSZSV62VNZJO/graph.json","fetch_events":"https://pith.science/api/pith-number/6PICN2B2K624T4CSZSV62VNZJO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO/action/storage_attestation","attest_author":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO/action/author_attestation","sign_citation":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO/action/citation_signature","submit_replication":"https://pith.science/pith/6PICN2B2K624T4CSZSV62VNZJO/action/replication_record"}},"created_at":"2026-07-05T09:23:11.213745+00:00","updated_at":"2026-07-05T09:23:11.213745+00:00"}