{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6DLLTUERI7WVEMQHO6PX3S7IWK","short_pith_number":"pith:6DLLTUER","schema_version":"1.0","canonical_sha256":"f0d6b9d09147ed523207779f7dcbe8b2a7acca1fe198e64b8f2c52be865e73d1","source":{"kind":"arxiv","id":"2306.03509","version":1},"attestation_state":"computed","paper":{"title":"Mega-TTS: Zero-Shot Text-to-Speech at Scale with Intrinsic Inductive Bias","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chen Zhang, Chunfeng Wang, Jinglin Liu, Qian Yang, Rongjie Huang, Shengpeng Ji, Xiang Yin, Yi Ren, Zejun Ma, Zhenhui Ye, Zhou Zhao, Ziyue Jiang","submitted_at":"2023-06-06T08:54:49Z","abstract_excerpt":"Scaling text-to-speech to a large and wild dataset has been proven to be highly effective in achieving timbre and speech style generalization, particularly in zero-shot TTS. However, previous works usually encode speech into latent using audio codec and use autoregressive language models or diffusion models to generate it, which ignores the intrinsic nature of speech and may lead to inferior or uncontrollable results. We argue that speech can be decomposed into several attributes (e.g., content, timbre, prosody, and phase) and each of them should be modeled using a module with appropriate indu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.03509","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-06-06T08:54:49Z","cross_cats_sorted":["cs.AI","cs.SD"],"title_canon_sha256":"f1f1c4b9ea2d851067c0a79e97b573dd4ab4306bdfbf1de968e8662728d79b15","abstract_canon_sha256":"dff8fe681484360e915f07cae00b6ef45a3aa55b06d7d480c27b37d66de5a479"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:58.947721Z","signature_b64":"v7zsfA+GvnGIYy1UATmQS4XcbfKxbIyV3czHQ7mr9Ke+no1i/TveO5sStg49WgNTXqqjHcwmcm0gdVbKs3S4Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f0d6b9d09147ed523207779f7dcbe8b2a7acca1fe198e64b8f2c52be865e73d1","last_reissued_at":"2026-07-05T06:17:58.947215Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:58.947215Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mega-TTS: Zero-Shot Text-to-Speech at Scale with Intrinsic Inductive Bias","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chen Zhang, Chunfeng Wang, Jinglin Liu, Qian Yang, Rongjie Huang, Shengpeng Ji, Xiang Yin, Yi Ren, Zejun Ma, Zhenhui Ye, Zhou Zhao, Ziyue Jiang","submitted_at":"2023-06-06T08:54:49Z","abstract_excerpt":"Scaling text-to-speech to a large and wild dataset has been proven to be highly effective in achieving timbre and speech style generalization, particularly in zero-shot TTS. However, previous works usually encode speech into latent using audio codec and use autoregressive language models or diffusion models to generate it, which ignores the intrinsic nature of speech and may lead to inferior or uncontrollable results. We argue that speech can be decomposed into several attributes (e.g., content, timbre, prosody, and phase) and each of them should be modeled using a module with appropriate indu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.03509","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.03509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.03509","created_at":"2026-07-05T06:17:58.947279+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.03509v1","created_at":"2026-07-05T06:17:58.947279+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.03509","created_at":"2026-07-05T06:17:58.947279+00:00"},{"alias_kind":"pith_short_12","alias_value":"6DLLTUERI7WV","created_at":"2026-07-05T06:17:58.947279+00:00"},{"alias_kind":"pith_short_16","alias_value":"6DLLTUERI7WVEMQH","created_at":"2026-07-05T06:17:58.947279+00:00"},{"alias_kind":"pith_short_8","alias_value":"6DLLTUER","created_at":"2026-07-05T06:17:58.947279+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21888","citing_title":"ProsoCodec: Prosody-Oriented Speech Codec for Voice Conversion","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01537","citing_title":"Two-Dimensional Quantization for Geometry-Aware Audio Coding","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02430","citing_title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK","json":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK.json","graph_json":"https://pith.science/api/pith-number/6DLLTUERI7WVEMQHO6PX3S7IWK/graph.json","events_json":"https://pith.science/api/pith-number/6DLLTUERI7WVEMQHO6PX3S7IWK/events.json","paper":"https://pith.science/paper/6DLLTUER"},"agent_actions":{"view_html":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK","download_json":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK.json","view_paper":"https://pith.science/paper/6DLLTUER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.03509&json=true","fetch_graph":"https://pith.science/api/pith-number/6DLLTUERI7WVEMQHO6PX3S7IWK/graph.json","fetch_events":"https://pith.science/api/pith-number/6DLLTUERI7WVEMQHO6PX3S7IWK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK/action/storage_attestation","attest_author":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK/action/author_attestation","sign_citation":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK/action/citation_signature","submit_replication":"https://pith.science/pith/6DLLTUERI7WVEMQHO6PX3S7IWK/action/replication_record"}},"created_at":"2026-07-05T06:17:58.947279+00:00","updated_at":"2026-07-05T06:17:58.947279+00:00"}