{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZDIOIGTMHODBP4E2ODXY3DEM2O","short_pith_number":"pith:ZDIOIGTM","schema_version":"1.0","canonical_sha256":"c8d0e41a6c3b8617f09a70ef8d8c8cd392653bafb19cd26061f8986fad6af4ca","source":{"kind":"arxiv","id":"2312.01479","version":6},"attestation_state":"computed","paper":{"title":"OpenVoice: Versatile Instant Voice Cloning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Wenliang Zhao, Xin Sun, Xumin Yu, Zengyi Qin","submitted_at":"2023-12-03T18:41:54Z","abstract_excerpt":"We introduce OpenVoice, a versatile voice cloning approach that requires only a short audio clip from the reference speaker to replicate their voice and generate speech in multiple languages. OpenVoice represents a significant advancement in addressing the following open challenges in the field: 1) Flexible Voice Style Control. OpenVoice enables granular control over voice styles, including emotion, accent, rhythm, pauses, and intonation, in addition to replicating the tone color of the reference speaker. The voice styles are not directly copied from and constrained by the style of the referen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.01479","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SD","submitted_at":"2023-12-03T18:41:54Z","cross_cats_sorted":["cs.LG","eess.AS"],"title_canon_sha256":"6807befab9465a67e9b1ac06daf577e02b8a2746a612dd5268ad3e148b6f381a","abstract_canon_sha256":"ae830671786a7706f500c4fd6794fdc5f558d861ccdb56cd2df6cd3431f9bff1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:15.500108Z","signature_b64":"+Y6UNTLsyzCyzkFCPVVOJ8TKtL84aTzBe/O9ebV4iOhlPecgDoaK7iO2LMOU65Rj65lnosfkWDtX29Wqz3WdDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8d0e41a6c3b8617f09a70ef8d8c8cd392653bafb19cd26061f8986fad6af4ca","last_reissued_at":"2026-07-05T08:56:15.499680Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:15.499680Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenVoice: Versatile Instant Voice Cloning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","eess.AS"],"primary_cat":"cs.SD","authors_text":"Wenliang Zhao, Xin Sun, Xumin Yu, Zengyi Qin","submitted_at":"2023-12-03T18:41:54Z","abstract_excerpt":"We introduce OpenVoice, a versatile voice cloning approach that requires only a short audio clip from the reference speaker to replicate their voice and generate speech in multiple languages. OpenVoice represents a significant advancement in addressing the following open challenges in the field: 1) Flexible Voice Style Control. OpenVoice enables granular control over voice styles, including emotion, accent, rhythm, pauses, and intonation, in addition to replicating the tone color of the reference speaker. The voice styles are not directly copied from and constrained by the style of the referen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.01479","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.01479/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.01479","created_at":"2026-07-05T08:56:15.499738+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.01479v6","created_at":"2026-07-05T08:56:15.499738+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.01479","created_at":"2026-07-05T08:56:15.499738+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZDIOIGTMHODB","created_at":"2026-07-05T08:56:15.499738+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZDIOIGTMHODBP4E2","created_at":"2026-07-05T08:56:15.499738+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZDIOIGTM","created_at":"2026-07-05T08:56:15.499738+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21343","citing_title":"An Evaluation Framework for Text-to-Speech Voice Reconstruction","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20218","citing_title":"Zero-VC: Zero-Lookahead Streaming Voice Conversion via Speaker Anonymization","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00726","citing_title":"AV-SyncBench: Decoupled Benchmarking of Temporal and Semantic Audio-Visual Synchronization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26136","citing_title":"Eroding Trust in Real Speech: A Large-Scale Study of Human Audio Deepfake Perception","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2510.19414","citing_title":"EchoFake: A Replay-Aware Dataset for Practical Speech Deepfake Detection","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22143","citing_title":"JUST-DUB-IT: Video Dubbing via Joint Audio-Visual Diffusion","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2412.10117","citing_title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12310","citing_title":"Poly-SVC: Polyphony-Aware Singing Voice Conversion with Harmonic Modeling","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23703","citing_title":"Talking Slide Avatars: Open-Source Multimodal Communication Approach for Teaching","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24794","citing_title":"V.O.I.C.E (Voice, Ownership, Identity, Control, Expression): Risk Taxonomy of Synthetic Voice Generation From Empirical Data","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11552","citing_title":"MimicLM: Zero-Shot Voice Imitation through Autoregressive Modeling of Pseudo-Parallel Speech Corpora","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12456","citing_title":"X-VC: Zero-shot Streaming Voice Conversion in Codec Space","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24770","citing_title":"Elderly-Contextual Data Augmentation via Speech Synthesis for Elderly ASR","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O","json":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O.json","graph_json":"https://pith.science/api/pith-number/ZDIOIGTMHODBP4E2ODXY3DEM2O/graph.json","events_json":"https://pith.science/api/pith-number/ZDIOIGTMHODBP4E2ODXY3DEM2O/events.json","paper":"https://pith.science/paper/ZDIOIGTM"},"agent_actions":{"view_html":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O","download_json":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O.json","view_paper":"https://pith.science/paper/ZDIOIGTM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.01479&json=true","fetch_graph":"https://pith.science/api/pith-number/ZDIOIGTMHODBP4E2ODXY3DEM2O/graph.json","fetch_events":"https://pith.science/api/pith-number/ZDIOIGTMHODBP4E2ODXY3DEM2O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O/action/storage_attestation","attest_author":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O/action/author_attestation","sign_citation":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O/action/citation_signature","submit_replication":"https://pith.science/pith/ZDIOIGTMHODBP4E2ODXY3DEM2O/action/replication_record"}},"created_at":"2026-07-05T08:56:15.499738+00:00","updated_at":"2026-07-05T08:56:15.499738+00:00"}