{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UJWKIH7U46OJK3D2WLJ6MDJ2IS","short_pith_number":"pith:UJWKIH7U","schema_version":"1.0","canonical_sha256":"a26ca41ff4e79c956c7ab2d3e60d3a44b752f4c2e79499f1584f0d8919af6909","source":{"kind":"arxiv","id":"2409.13910","version":1},"attestation_state":"computed","paper":{"title":"Zero-shot Cross-lingual Voice Transfer for TTS","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Andrew Rosenberg, Bhuvana Ramabhadran, Fadi Biadsy, Gary Wang, Isaac Elias, Kyle Kastner, Youzheng Chen","submitted_at":"2024-09-20T21:27:13Z","abstract_excerpt":"In this paper, we introduce a zero-shot Voice Transfer (VT) module that can be seamlessly integrated into a multi-lingual Text-to-speech (TTS) system to transfer an individual's voice across languages. Our proposed VT module comprises a speaker-encoder that processes reference speech, a bottleneck layer, and residual adapters, connected to preexisting TTS layers. We compare the performance of various configurations of these components and report Mean Opinion Score (MOS) and Speaker Similarity across languages. Using a single English reference speech per speaker, we achieve an average voice tra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13910","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-09-20T21:27:13Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"35ec36044bfdc41bbf87104edb91a5e961dd8c17eedfc591a952d513cdf16881","abstract_canon_sha256":"b96d3e1f0cadc86b457d2b8770f2a184ded3f2942d3c1a58f2d96a2fb972d41a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:08.664269Z","signature_b64":"fDzHHLtHGuH8ro/NX94TMPX/KsPhVXG3uPXKcxlpjlp9wbkgKKNhxYxP87lEkAT7d8y+h2ULgpOwDzIFfCDWCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a26ca41ff4e79c956c7ab2d3e60d3a44b752f4c2e79499f1584f0d8919af6909","last_reissued_at":"2026-07-05T09:10:08.663794Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:08.663794Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-shot Cross-lingual Voice Transfer for TTS","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Andrew Rosenberg, Bhuvana Ramabhadran, Fadi Biadsy, Gary Wang, Isaac Elias, Kyle Kastner, Youzheng Chen","submitted_at":"2024-09-20T21:27:13Z","abstract_excerpt":"In this paper, we introduce a zero-shot Voice Transfer (VT) module that can be seamlessly integrated into a multi-lingual Text-to-speech (TTS) system to transfer an individual's voice across languages. Our proposed VT module comprises a speaker-encoder that processes reference speech, a bottleneck layer, and residual adapters, connected to preexisting TTS layers. We compare the performance of various configurations of these components and report Mean Opinion Score (MOS) and Speaker Similarity across languages. Using a single English reference speech per speaker, we achieve an average voice tra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13910","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13910/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13910","created_at":"2026-07-05T09:10:08.663853+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13910v1","created_at":"2026-07-05T09:10:08.663853+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13910","created_at":"2026-07-05T09:10:08.663853+00:00"},{"alias_kind":"pith_short_12","alias_value":"UJWKIH7U46OJ","created_at":"2026-07-05T09:10:08.663853+00:00"},{"alias_kind":"pith_short_16","alias_value":"UJWKIH7U46OJK3D2","created_at":"2026-07-05T09:10:08.663853+00:00"},{"alias_kind":"pith_short_8","alias_value":"UJWKIH7U","created_at":"2026-07-05T09:10:08.663853+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.07302","citing_title":"XEmoRAG: Cross-Lingual Emotion Transfer with Controllable Intensity Using Retrieval-Augmented Generation","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS","json":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS.json","graph_json":"https://pith.science/api/pith-number/UJWKIH7U46OJK3D2WLJ6MDJ2IS/graph.json","events_json":"https://pith.science/api/pith-number/UJWKIH7U46OJK3D2WLJ6MDJ2IS/events.json","paper":"https://pith.science/paper/UJWKIH7U"},"agent_actions":{"view_html":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS","download_json":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS.json","view_paper":"https://pith.science/paper/UJWKIH7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13910&json=true","fetch_graph":"https://pith.science/api/pith-number/UJWKIH7U46OJK3D2WLJ6MDJ2IS/graph.json","fetch_events":"https://pith.science/api/pith-number/UJWKIH7U46OJK3D2WLJ6MDJ2IS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS/action/storage_attestation","attest_author":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS/action/author_attestation","sign_citation":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS/action/citation_signature","submit_replication":"https://pith.science/pith/UJWKIH7U46OJK3D2WLJ6MDJ2IS/action/replication_record"}},"created_at":"2026-07-05T09:10:08.663853+00:00","updated_at":"2026-07-05T09:10:08.663853+00:00"}