{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Z5LK2EJV2SMPNDALJUT7FZWIVV","short_pith_number":"pith:Z5LK2EJV","schema_version":"1.0","canonical_sha256":"cf56ad1135d498f68c0b4d27f2e6c8ad5d0100fb689e59316ffdf985e3369252","source":{"kind":"arxiv","id":"2305.19269","version":1},"attestation_state":"computed","paper":{"title":"Make-A-Voice: Unified Voice Synthesis With Discrete Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chao Weng, Chunlei Zhang, Dongchao Yang, Dong Yu, Luping Liu, Rongjie Huang, Yongqi Wang, Zhenhui Ye, Zhou Zhao, Ziyue Jiang","submitted_at":"2023-05-30T17:59:26Z","abstract_excerpt":"Various applications of voice synthesis have been developed independently despite the fact that they generate \"voice\" as output in common. In addition, the majority of voice synthesis models currently rely on annotated audio data, but it is crucial to scale them to self-supervised datasets in order to effectively capture the wide range of acoustic variations present in human voice, including speaker identity, emotion, and prosody. In this work, we propose Make-A-Voice, a unified framework for synthesizing and manipulating voice signals from discrete representations. Make-A-Voice leverages a \"c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.19269","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2023-05-30T17:59:26Z","cross_cats_sorted":["cs.AI","cs.CL","cs.SD"],"title_canon_sha256":"4c55a57a3c335ce993a4899c0b7f02a7224dcfe07ab7c838c9bf2f4f78021d26","abstract_canon_sha256":"b6f0e9c5bafba085bfa7526ec52b49f6d0401a51d05c56791af9ebe1001b8ad3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:15:40.304993Z","signature_b64":"Yx/WLouOIAys6+Ic69aYEq3eSiZeaNZBdPO08BfO13Ys840Od/aWc9Dw8HYDQq7YgVkrx28K/R6GPm4kqJi1DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf56ad1135d498f68c0b4d27f2e6c8ad5d0100fb689e59316ffdf985e3369252","last_reissued_at":"2026-07-05T06:15:40.304482Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:15:40.304482Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Make-A-Voice: Unified Voice Synthesis With Discrete Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Chao Weng, Chunlei Zhang, Dongchao Yang, Dong Yu, Luping Liu, Rongjie Huang, Yongqi Wang, Zhenhui Ye, Zhou Zhao, Ziyue Jiang","submitted_at":"2023-05-30T17:59:26Z","abstract_excerpt":"Various applications of voice synthesis have been developed independently despite the fact that they generate \"voice\" as output in common. In addition, the majority of voice synthesis models currently rely on annotated audio data, but it is crucial to scale them to self-supervised datasets in order to effectively capture the wide range of acoustic variations present in human voice, including speaker identity, emotion, and prosody. In this work, we propose Make-A-Voice, a unified framework for synthesizing and manipulating voice signals from discrete representations. Make-A-Voice leverages a \"c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.19269","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.19269/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.19269","created_at":"2026-07-05T06:15:40.304549+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.19269v1","created_at":"2026-07-05T06:15:40.304549+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.19269","created_at":"2026-07-05T06:15:40.304549+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z5LK2EJV2SMP","created_at":"2026-07-05T06:15:40.304549+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z5LK2EJV2SMPNDAL","created_at":"2026-07-05T06:15:40.304549+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z5LK2EJV","created_at":"2026-07-05T06:15:40.304549+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":239,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01677","citing_title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19883","citing_title":"CoMelSinger: Discrete Token-Based Zero-Shot Singing Synthesis With Structured Melody Control and Guidance","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV","json":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV.json","graph_json":"https://pith.science/api/pith-number/Z5LK2EJV2SMPNDALJUT7FZWIVV/graph.json","events_json":"https://pith.science/api/pith-number/Z5LK2EJV2SMPNDALJUT7FZWIVV/events.json","paper":"https://pith.science/paper/Z5LK2EJV"},"agent_actions":{"view_html":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV","download_json":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV.json","view_paper":"https://pith.science/paper/Z5LK2EJV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.19269&json=true","fetch_graph":"https://pith.science/api/pith-number/Z5LK2EJV2SMPNDALJUT7FZWIVV/graph.json","fetch_events":"https://pith.science/api/pith-number/Z5LK2EJV2SMPNDALJUT7FZWIVV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV/action/storage_attestation","attest_author":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV/action/author_attestation","sign_citation":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV/action/citation_signature","submit_replication":"https://pith.science/pith/Z5LK2EJV2SMPNDALJUT7FZWIVV/action/replication_record"}},"created_at":"2026-07-05T06:15:40.304549+00:00","updated_at":"2026-07-05T06:15:40.304549+00:00"}