{"as_of":"2026-08-12T22:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5290f88f55849b2689842af70f87bf356014c39838c51d69aeca0dc92ec8a9c4","coverage":[{"denominator":75,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":75,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T23:41:01.512970Z","state":"measured"},{"denominator":75,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":75,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.20048/citation-record","integrity":"/paper/2412.20048/integrity","json":"/paper/2412.20048/citation-record.json","paper":"/paper/2412.20048"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.262963Z","title":"The amazing benefits of being bilingual,","venue":null,"work_id":"d89cccc3-43fc-4b9b-9bce-d14dfc0e117c","year":2016},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.917403Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:58e3c02e44402994f07830de626bfa04a5f67974d3a4ec92af4d1a720b070af1","observation_id":"c23cb365-eee1-4a1b-bb8a-07ca5bd05705","resolution":{"observed_at":"2026-08-10T23:41:03.269476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.236440Z","title":"Disentangled representation learning for multilingual speaker recogni- tion,","venue":null,"work_id":"bc9d71e4-763f-4216-913b-39db383ce612","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.923994Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:15af93e168e4d55c756077d09e1aed55bcc4d3b850aa1eebf60de42626e49003","observation_id":"5d15a981-3d6f-4e69-a3a7-70b09b9846fd","resolution":{"observed_at":"2026-08-10T23:41:03.242873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.214000Z","title":"Crosslingual and multilingual speech recognition based on the speech manifold,","venue":null,"work_id":"9dfa38d3-67b5-4493-8503-65f02277cf18","year":null},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.932355Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:7e5d5af064a0babfc1dbca2b334b30e3f734574737f54190b57f2d781d7cc091","observation_id":"8216c3ff-4d1a-4620-a9b8-fd20ab69d79a","resolution":{"observed_at":"2026-08-10T23:41:03.220288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.193890Z","title":"Distilling a pretrained language model to a multilingual asr model,","venue":null,"work_id":"3baa0fcf-ad01-461e-b9c0-04a3edb53d4c","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.938640Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:1aefd9b04513562a5067737a4c09c3abe6daf3ee8cc26a5099f8c4692e628d38","observation_id":"2b16526b-174f-48db-b2f1-7ea9f1eaa3fc","resolution":{"observed_at":"2026-08-10T23:41:03.199694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.168055Z","title":"Joint ASR and language identification using RNN-T: An efficient approach to dynamic language switching,","venue":null,"work_id":"9b014a4f-119e-4203-bdd7-6e163d0ae5f9","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.946711Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:b3baccbf24a1c7e4d350aeddd4e26fb38fe84a5c8f870eb4b97b208603e9fd12","observation_id":"2b79a606-c8fd-4e75-98e3-43b4e84b1814","resolution":{"observed_at":"2026-08-10T23:41:03.174204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.144654Z","title":"Joint unsupervised and supervised learning for context-aware language iden- tification,","venue":null,"work_id":"5da52a8a-a540-47e7-8515-64aa8a6e1969","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.953481Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:45c68988f0ecb829c1d76060014eb6bc3243f4745178246ef43113c284ed2c51","observation_id":"eed4ff27-cd21-4963-ba27-574002257665","resolution":{"observed_at":"2026-08-10T23:41:03.151563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.125146Z","title":"Fastpitch: Parallel text-to-speech with pitch prediction,","venue":null,"work_id":"a5c365c2-a179-4c49-a86d-efc58b72e243","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.961403Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:f6c3bf8be0bdfacc24082760159785a8332eaee5a8fb0c80c91f6bf8b1ab4d85","observation_id":"349dcebc-a7fa-433e-97cd-28dc3958c96b","resolution":{"observed_at":"2026-08-10T23:41:03.131618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.100982Z","title":"Speaker adaptive text-to-speech with timbre-normalized vector-quantized feature,","venue":null,"work_id":"eb625625-8b2d-4f9f-bdae-f779d7185bc9","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.969012Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:4187ea733c42f793b03f1d3688ec6ef5bbff3ac215613315405d899a6ff5b691","observation_id":"bdb4b98c-6964-4637-9bcc-d30e1d0d7339","resolution":{"observed_at":"2026-08-10T23:41:03.107774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.078214Z","title":"EfficientTTS 2: Variational end-to-end text-to-speech synthesis and voice conversion,","venue":null,"work_id":"eef167cc-5325-4d5f-9114-a6ebebe53501","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.976039Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:ef832b0211294511126bfc31701d7945efbc9305c98ada435b9b012a785e4611","observation_id":"a604f3c6-c4ef-4ebd-896c-415750cedae8","resolution":{"observed_at":"2026-08-10T23:41:03.085372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.053282Z","title":"Bytes are all you need: End-to-end multilingual speech recognition and synthesis with bytes,","venue":null,"work_id":"2418e765-5870-4a77-a5ae-137d1e1237e1","year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.982674Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:1cfc68bd841009eb7968d0040d7b9d61bee5405e6b02264cab779b7e32106c97","observation_id":"05260672-3867-48f6-89a2-bf5c14eae1b5","resolution":{"observed_at":"2026-08-10T23:41:03.059788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.031777Z","title":"Improve cross-lingual text-to- speech synthesis on monolingual corpora with pitch contour informa- tion,","venue":null,"work_id":"7071bfd1-3aa2-4258-a5be-bb9bf4aa587d","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.989110Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:9e3c360eaac0843333955f289dc537b6079d1ff9d0807449687e4d02048a2b70","observation_id":"a11dc593-9a18-48e6-9cf7-05aff151bfdd","resolution":{"observed_at":"2026-08-10T23:41:03.039156Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:03.009420Z","title":"Language-agnostic meta-learning for low-resource text-to-speech with articulatory features,","venue":null,"work_id":"4ae13d63-e3e9-4f81-ae82-33ad457bc340","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:00.994932Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:f88fa5c17d3ae8108894cde937b86d956c74fade4a977201f744d539b525d71b","observation_id":"8302a362-5913-4d46-8a29-daf86d124842","resolution":{"observed_at":"2026-08-10T23:41:03.015542Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.988284Z","title":"Learning to speak fluently in a foreign language: Multilingual speech synthesis and cross-language voice cloning,","venue":null,"work_id":"cddeaccf-fecd-4ad2-9e87-a1f691907674","year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.004849Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:d4448fb5d87aeabf1c3998ae7ce3c7e98063c1550fd639e18dadf9e02cc78273","observation_id":"af47961d-8ebb-45e2-a4ea-6e4228fde64b","resolution":{"observed_at":"2026-08-10T23:41:02.995913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.967683Z","title":"Disentan- gled speaker and language representations using mutual information minimization and domain adaptation for cross-lingual TTS,","venue":null,"work_id":"c9287005-5338-40d0-bff7-890bb5604b33","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.013367Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:040279712ec05f169674a9ae01f55caea3d2d7993345620afb00acc0d13bd331","observation_id":"8d41d9eb-99b6-4812-a974-2a26a56b0778","resolution":{"observed_at":"2026-08-10T23:41:02.973107Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.949608Z","title":"GenerTTS: Pronunciation disentanglement for timbre and style generalization in cross-lingual text-to-speech,","venue":null,"work_id":"69b9927c-43af-4556-ba14-4160326a0d10","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.020861Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:6bab2f179ad2ea0791f4de6bf49dfa458e4d443a9c7d3d052f2870a96ec7f7de","observation_id":"c6f464f7-6ff1-4529-8ab7-2d9d93c1e1a1","resolution":{"observed_at":"2026-08-10T23:41:02.954486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.922003Z","title":"DSE-TTS: Dual speaker embedding for cross-lingual text-to-speech,","venue":null,"work_id":"f3cbec09-6330-4ef0-b2fb-dd5931376d4e","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.026896Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:023b487920c13f0f5a0113d3615272412b0d02390e70ed44b1272c90db3e0c51","observation_id":"cd8acda7-6487-45a4-82f8-9eba3e9e4432","resolution":{"observed_at":"2026-08-10T23:41:02.929948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14398","last_updated":"2024-08-27T03:19:09Z","snapshot_observed_at":"2026-07-06T17:06:54.562888Z","submitted_at":"2023-12-22T02:54:15Z","title":"ZMM-TTS: Zero-shot Multilingual and Multispeaker Speech Synthesis Conditioned on Self-supervised Discrete Speech Representations","version":2},"cited_work":{"arxiv_id":"2312.14398","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.14398","snapshot_observed_at":"2026-08-10T23:41:01.678270Z","title":"ZMM-TTS: Zero-shot Multilingual and Multispeaker Speech Synthesis Conditioned on Self-supervised Discrete Speech Representations","venue":"cs.SD","work_id":"8d7ee1b9-2dd1-4c61-a602-aaa0da886fbb","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.037296Z"},"links":{"cited_paper":"/paper/2312.14398","citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:5d84f36c56cceeb274c0016e2c9731ad80f9588fce44f596c05a38294a16b8d8","observation_id":"93e48c2c-23e7-41ff-8d96-4703c38e8ec7","resolution":{"observed_at":"2026-08-10T23:41:01.687794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.901843Z","title":"Unit selection in a concatenative speech synthesis system using a large speech database,","venue":null,"work_id":"3b64d761-8b30-4037-b893-09f4ab65c739","year":1996},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.048078Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:8ddf63a4229dd946291d5f8ddbbba069c6a760ec8c378c185b9eb4e76e1f3f71","observation_id":"24b0fff1-493f-4e57-bbd5-7cbf40f7e594","resolution":{"observed_at":"2026-08-10T23:41:02.907351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.876335Z","title":"Statistical parametric speech synthesis,","venue":null,"work_id":"a618a334-5839-420c-8392-e6adaa89d1cd","year":2007},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.054553Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:4b2bff779027e7398ea0ec4bb88673d0b1da4b738b22acefdc35b0bef5d27d31","observation_id":"91598463-fd05-41fc-9cb8-7aa59d3722e9","resolution":{"observed_at":"2026-08-10T23:41:02.884267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.858483Z","title":"Naturalspeech: End-to-end text-to-speech synthesis with human-level quality,","venue":null,"work_id":"f7b23633-0144-4b4f-8997-15d14b8ff666","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.061028Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:9ebd0fed809b56c8b821463277fa26f638c9c77e79d3a8a41cd0c990fa32a594","observation_id":"6d1d7ac2-cff8-4500-a8f4-be36d07032a7","resolution":{"observed_at":"2026-08-10T23:41:02.864249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.838578Z","title":"Harmonic-net: Fundamental frequency and speech rate controllable fast neural vocoder,","venue":null,"work_id":"b036fee8-1db5-4b59-a8a2-a439f4da537e","year":1902},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.067703Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:276e26fda03f1d0adcdf0de91de9153f4f940e8ccc6d0b360eabe838eef8ff11","observation_id":"30070c5f-f1c9-4403-9a58-cf8fa98d6ec9","resolution":{"observed_at":"2026-08-10T23:41:02.845442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.813192Z","title":"Fregrad: Lightweight and fast frequency-aware diffusion vocoder,","venue":null,"work_id":"50f02d8c-72c7-474e-924c-217d36384c67","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.074760Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:97a1c96614edc3aee12c463312e6bc0cf1de6993244ecfa604726354f5cdb4de","observation_id":"17f12e01-95b9-44b2-8e78-61efbc5e51c5","resolution":{"observed_at":"2026-08-10T23:41:02.824269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.794353Z","title":"TriniTTS: Pitch-controllable end-to-end TTS without external aligner","venue":null,"work_id":"02c3ed17-9c32-44b0-ba38-8014eec50776","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.082362Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:1388abe322bdd808c8d77ed2332a2d0f213d9c96d0d0a3cede4d44810118cb9f","observation_id":"5b22e679-ee23-42cb-bb09-a32f3753b78f","resolution":{"observed_at":"2026-08-10T23:41:02.800273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.776422Z","title":"Hierspeech: Bridging the gap between text and speech by hierarchical variational inference using self-supervised representations for speech synthesis,","venue":null,"work_id":"d4bb5aed-bebd-4046-82c0-5eeccaa486b1","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.087731Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:894a67035f9bc07260fada45df33e7af3937393250ac36e430ea917e6e4d3bc1","observation_id":"86d5654b-fd8c-4a89-b268-c97125f338ff","resolution":{"observed_at":"2026-08-10T23:41:02.782213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1609.03499","last_updated":"2016-09-19T18:04:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-09-12T17:29:40Z","title":"WaveNet: A Generative Model for Raw Audio","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.03499","snapshot_observed_at":"2026-08-10T23:41:01.095472Z","title":"Wavenet: A gener- ative model for raw audio,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.095472Z"},"links":{"cited_paper":"/paper/1609.03499","citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:08845f0675e1f0613be2aed6c210a722198fa199991555b485097f9961ca6c2d","observation_id":"614d5c98-5ce2-4392-8be2-08ff8d25809c","resolution":{"observed_at":"2026-08-10T23:41:01.095472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.757721Z","title":"Deep voice: Real-time neural text-to-speech,","venue":null,"work_id":"a01eb754-88a7-4383-ac37-dd950f437b4d","year":2017},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.105039Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:cc37088fa815c92b4981ed0c1c5d295650760d8005103cb69e90e0df2bf1d56d","observation_id":"4bceb40e-42d5-4061-a722-705378184bec","resolution":{"observed_at":"2026-08-10T23:41:02.763678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.734948Z","title":"Natural TTS synthesis by conditioning wavenet on mel spectrogram predictions,","venue":null,"work_id":"98a31153-8365-4b90-92d0-4cbcc88ff004","year":2018},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.116179Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:afa41cc260e8f3951aa070a46e37e7915f7b3037b80166380abaea0ea3ab1d4e","observation_id":"c9318edb-cb90-4bc6-b523-4e1e1278648e","resolution":{"observed_at":"2026-08-10T23:41:02.740586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.716157Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":"16a54e22-d133-44cc-af70-f0b619cdc746","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.123118Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:0e644666845e78ded718c74ced8eb306153ca5ee56eac5af25750a4807752e48","observation_id":"57adf011-d731-4c2d-8505-c9b089076f6b","resolution":{"observed_at":"2026-08-10T23:41:02.722246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.698049Z","title":"Matcha- TTS: A fast TTS architecture with conditional flow matching,","venue":null,"work_id":"a3bb2637-5921-4160-8ce3-106424a4228b","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.128890Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:f9040fdfb0aa7f4b6448ef3c808a1a898491b9b1df50454e2e5447ca4483fadd","observation_id":"199ac34f-19af-4db2-a874-85977630b03b","resolution":{"observed_at":"2026-08-10T23:41:02.703668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.680111Z","title":"Multispeech: Multi-speaker text to speech with transformer,","venue":null,"work_id":"148dc260-6fd9-4eb0-a989-ecf72e0c68c1","year":2020},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.135253Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:0d17ea46bd9099d661deff631d278e20bfbca53ee36f15b64245616a2d229d4e","observation_id":"bb6e89df-9fc1-48d8-ac11-f7b31e9a642c","resolution":{"observed_at":"2026-08-10T23:41:02.685786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.661614Z","title":"Lightspeech: Lightweight and fast text to speech with neural architecture search,","venue":null,"work_id":"5e87a381-0f2e-4600-8d81-b8ee88a4a8b9","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.141137Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:e339bd4fce6d1be724521e763e86a12600c60483e613b2a0010b89592e7d5093","observation_id":"a10192b2-9bf5-40e9-b767-bef9dbf8dbc8","resolution":{"observed_at":"2026-08-10T23:41:02.666778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.640572Z","title":"Phonological features for 0-shot multilingual speech synthesis,","venue":null,"work_id":"b8b2fa46-201e-429f-b17c-ebf603bf2df0","year":2020},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.147694Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:5c7f865bd2d1019c78009f6a77a716729c90fe9943e3249a32aa5277f0474e5a","observation_id":"92c323f5-604a-4040-ab6b-ff61414a0bcd","resolution":{"observed_at":"2026-08-10T23:41:02.646583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.618575Z","title":"Text-inductive graphone-based language adaptation for low-resource speech synthesis,","venue":null,"work_id":"64f1f0ad-608e-45f8-9c53-aa74b924db6f","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.153949Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:ea70c4e014c49865938de3b4f0f24795f218f0756597864379c0befaa13dd7b1","observation_id":"85d6b405-fd35-4a37-adff-08282372a955","resolution":{"observed_at":"2026-08-10T23:41:02.626269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.593495Z","title":"BERT: Pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"793ee04b-aa84-4974-a7a3-c944ad85da8e","year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.163504Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:a7109ef4489f842ae99ace53a9702c1bcd7387702e98cdc635d64f948fb93230","observation_id":"7e7c046c-2cf5-42ac-b3f2-7cfea3e0e140","resolution":{"observed_at":"2026-08-10T23:41:02.601319Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.572529Z","title":"Domain-adversarial training of neural networks,","venue":null,"work_id":"813d81ba-0824-4850-beb1-3ed27d34a1c7","year":2016},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.171886Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:7bef3084aa34e597f1f25a9eb76e4c052b5ab863af063fe7ff78d7dab247b1e9","observation_id":"7af7583e-39dd-4fd3-acf9-644ea4cd4064","resolution":{"observed_at":"2026-08-10T23:41:02.579706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.520440Z","title":"Learning disentangled representations via mutual information estimation,","venue":null,"work_id":"42cbc4f6-3fa8-4a38-8962-7caa3239987d","year":2020},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.180485Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:3d09400606fedfe2e51388b5d88ef4b77a50893d49b0c6af7add71f46120bf73","observation_id":"785c3b1b-eafd-4333-8647-7bc47c845abd","resolution":{"observed_at":"2026-08-10T23:41:02.545882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.501609Z","title":"SANE-TTS: Stable and natural end-to-end multilingual text-to-speech,","venue":null,"work_id":"35439908-36c5-4b79-a0fe-013989f46189","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.190551Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:1a53b4577a3d44efde01577bd55d06962a18340aa7d283c6f2d5aed86a343ec8","observation_id":"36afed2e-9437-4b04-a4dc-48e4d2cfb84e","resolution":{"observed_at":"2026-08-10T23:41:02.506996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.484060Z","title":"Crossspeech: Speaker-independent acoustic representation for cross- lingual speech synthesis,","venue":null,"work_id":"d8a1ba45-ae60-48af-9f84-ace365c3ca13","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.198083Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:1e61d29bb29f1314e7efaf7bf32a5adb554ec64e731366352c6254eeb911352d","observation_id":"97c9a134-6f77-4bdc-b621-9e75747d8f33","resolution":{"observed_at":"2026-08-10T23:41:02.489429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.464671Z","title":"Invariant risk minimization,","venue":null,"work_id":"1d05a1c7-fcaf-44b3-a58f-9e617a73b6f8","year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.206074Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:09944421c6de25f70e9c810b88e11a36df6271e0a59b6fbe472c789781498a01","observation_id":"a47dc9af-30fa-4784-a4a9-756411a563de","resolution":{"observed_at":"2026-08-10T23:41:02.471418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.443108Z","title":"Domain generalization with mixstyle,","venue":null,"work_id":"67902108-b269-4b9e-a53f-f6de978673bc","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.213292Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:3d7734a5e8cfe9e84289bab7f1ecdcd0c104c2daa6d8a4529971d280c857919a","observation_id":"dc79d5a0-1db8-437b-8f03-c3ec0ddb7f7d","resolution":{"observed_at":"2026-08-10T23:41:02.451892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.421050Z","title":"One TTS alignment to rule them all,","venue":null,"work_id":"fc35f8b0-77a3-4faa-9f23-398ef22b08fa","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.232108Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:94c5ae774c9ea744dea54eafa6bf007cc458509338fd83256b998dae6b32cb8e","observation_id":"b1a2639b-98e0-4e28-95e7-d75dd4702376","resolution":{"observed_at":"2026-08-10T23:41:02.427354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.393196Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":"9ee8c76a-aebe-49df-8395-ae53962ad01b","year":2020},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.240834Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:72ff01e047a7d40b2af37c56cd896df806e43a756a4a25e67ba0d8e9d939896e","observation_id":"d0c5ad9d-09b4-4dde-967d-f6f1541e0898","resolution":{"observed_at":"2026-08-10T23:41:02.407581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.371374Z","title":"Feature-critic networks for heterogeneous domain generalization,","venue":null,"work_id":"dc635dca-c71e-4dfd-8403-a050a4d7a32c","year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.258299Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:28313bc2895a6fae569fa88997d007045463c7eed71b2e6432c39d56f85c4f39","observation_id":"3ba2697f-2546-41d8-ae46-6659e9c2772f","resolution":{"observed_at":"2026-08-10T23:41:02.378049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.342550Z","title":"Generspeech: Towards style transfer for generalizable out-of-domain text-to-speech,","venue":null,"work_id":"8c335084-b074-4594-9d04-9cd088ce1d7f","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.267937Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:05c59ef9f4002f2baddb93b977b79f5aa532bff29799329b8c5abec61f80a11d","observation_id":"a66522b6-11f1-4ab5-b738-483249044bb8","resolution":{"observed_at":"2026-08-10T23:41:02.349918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.321355Z","title":"PV AE-TTS: Adaptive text-to-speech via progressive style adaptation,","venue":null,"work_id":"56d90644-6391-417c-abe5-3ce6ba10bb14","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.275765Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:17609789dad9a65a37d427c89d2868622607e5fe25ec6bdf100769e3355e329e","observation_id":"c4ece250-45d7-4872-beb6-ebd9ae1e1680","resolution":{"observed_at":"2026-08-10T23:41:02.327619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.297098Z","title":"Style normalization and restitution for domain generalization and adaptation,","venue":null,"work_id":"d846f40d-b0a7-4871-b106-373e3400e9e5","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.282956Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:db2b5e62cde3ad50611faa43c10059a9ca7c56cdd6fa808d893dabc855074ff4","observation_id":"fcfee05f-68af-4a65-998e-6b01c8c3a593","resolution":{"observed_at":"2026-08-10T23:41:02.302559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.268619Z","title":"Prosospeech: Enhancing prosody with quantized vector pre-training in text-to-speech,","venue":null,"work_id":"731bfc67-87d5-4ff4-8ce5-b82b7d37f306","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.290550Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:e4508911b928aa555ae3832b901e8b6b3287def155f7c08d6a7af6194442b302","observation_id":"602d7743-9ae7-4ad6-b30e-d0c9a3de8ea6","resolution":{"observed_at":"2026-08-10T23:41:02.277338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.244384Z","title":"Diffprosody: Diffusion-based latent prosody generation for expressive speech synthesis with prosody conditional adversarial training,","venue":null,"work_id":"4ca090b5-9671-404c-9554-fa31d3f5d63a","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.296819Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:6353e646d46224410516116402d02d409c046f5100f958f3cf66ab4185429ad3","observation_id":"01b8547f-fdc7-467c-a438-a8e4011503cd","resolution":{"observed_at":"2026-08-10T23:41:02.252247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.221671Z","title":"pYIN: A fundamental frequency estimator using probabilistic threshold distributions,","venue":null,"work_id":"6ecb33da-280c-428e-9718-6c2eb68a8df3","year":2014},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.302355Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:17ebdc6c75354c7920e2abe03b62463275edc3cd39711bebca106258ebc9b6f9","observation_id":"26cfbfa1-825f-4daa-a197-d4e51f33f2c6","resolution":{"observed_at":"2026-08-10T23:41:02.228142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.203597Z","title":"Neural analysis and synthesis: Reconstructing speech from self-supervised representations,","venue":null,"work_id":"213abe4b-7402-49dd-ad56-c21c4aec918f","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.310006Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:1bbd6dd0c56af34422faab49fc51913b997a119e5e3a892b8b2d1ca4246b7fbe","observation_id":"f75cf353-7153-4d34-88dc-12db8c1bde0a","resolution":{"observed_at":"2026-08-10T23:41:02.209008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.183941Z","title":"Exploring wav2vec 2.0 on speaker verification and language identification,","venue":null,"work_id":"b81cb36d-af03-467a-99a1-9199c0b38422","year":2020},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.320135Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:4f748011f3bee1ccbf201e461eb128f74dc9577cc1d81447c76413c3f0905179","observation_id":"760857da-4f2d-4ad3-b8d5-19ec24273660","resolution":{"observed_at":"2026-08-10T23:41:02.190483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.162216Z","title":"Let there be sound: Reconstructing high quality speech from silent videos,","venue":null,"work_id":"f81b3cd1-4765-4755-889d-0b7e4f79d948","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.326985Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:90ab87f19f602d5176cbbfd5d6a5decb176357c1f1d9897a1a0f2f2d7a344ab5","observation_id":"71799840-ac50-45c1-893c-407a7d1fd3a2","resolution":{"observed_at":"2026-08-10T23:41:02.169374Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.136447Z","title":"Scaling speech technology to 1,000+ languages,","venue":null,"work_id":"2a8b0474-3f38-4ef3-b4b3-643a04246332","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.334513Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:a19b9625f951330a04d227d6c5dd0de460dccc9b6fc12c14c7bed43d9c8a8327","observation_id":"7cf0cd41-4470-4132-a999-417d4e9ebbae","resolution":{"observed_at":"2026-08-10T23:41:02.141990Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.117276Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":"a49d5b27-e45c-48b7-a79b-83ee0ff7abbc","year":2020},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.341458Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:f21584f97bb7101fe2ac8ea148f0569071268e0eb41f0effe8c560abb491c20c","observation_id":"d3ef05fd-f846-4cc0-a9b2-c146ec400b44","resolution":{"observed_at":"2026-08-10T23:41:02.123668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.095202Z","title":"Connection- ist temporal classification: labelling unsegmented sequence data with recurrent neural networks,","venue":null,"work_id":"935f1deb-d70a-4e02-afc3-23351dacee66","year":2006},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.347868Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:606ce93161a6e0fb41bcdbb5516fd19a5968924b326ea86cee966bbd759e7a90","observation_id":"0f4223bf-e28b-4eee-8e8c-47859fd98892","resolution":{"observed_at":"2026-08-10T23:41:02.101079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.074921Z","title":"The LJ speech dataset,","venue":null,"work_id":"ca8edbfa-cbe9-418c-9d85-7e33e7ad7267","year":2017},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.356498Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:7a9311decb3a3019b1997e493742057cc09dc167ee88a265064bc39684e13c8f","observation_id":"acaa876b-4f89-4927-9574-4d563ef18ac5","resolution":{"observed_at":"2026-08-10T23:41:02.082325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.363187Z","title":"CSTR VCTK corpus: English multi-speaker corpus for CSTR voice cloning toolkit,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.363187Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:c02606d966239167835a7670a8cb59a174517aa0ec6003def0f3d532649fa6b4","observation_id":"a57298f3-6db4-4c26-bd39-344a6ea1c185","resolution":{"observed_at":"2026-08-10T23:41:01.363187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.053805Z","title":"The BIAOBEI dataset,","venue":null,"work_id":"b9eef238-4792-4cd3-8758-92d5c01a05d3","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.373919Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:875028541100a712f29efe9ead8a1135a8fe39e9f53ed550c4de1b111f7590c3","observation_id":"f5c4ab24-b74c-4038-a6a3-f30616549ad2","resolution":{"observed_at":"2026-08-10T23:41:02.060733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.035519Z","title":"AISHELL-3: A multi- speaker Mandarin TTS corpus,","venue":null,"work_id":"8a38eafb-434e-47e4-bda8-db402c621a52","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.384569Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:62c558b98d40df6deb53ff696bd50cc842af5997d61c3533d45da2f30238b398","observation_id":"0e24fc1d-ff54-4eac-9a95-82c80d341ab7","resolution":{"observed_at":"2026-08-10T23:41:02.040412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:02.015437Z","title":"CSS10: A collection of single speaker speech datasets for 10 languages,","venue":null,"work_id":"f88c47c7-8dbb-4ec4-abf5-544e6e065c33","year":2019},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.393788Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:07c59a1776761212e7a511bebf1d4865a25be2109c1684e4564c64e209ac3c14","observation_id":"db8ce6fb-d726-4e9a-8b6f-d6c4797a14b3","resolution":{"observed_at":"2026-08-10T23:41:02.022899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.00354","last_updated":"2017-10-28T05:28:01Z","snapshot_observed_at":"2026-08-10T14:44:27.962270Z","submitted_at":"2017-10-28T05:28:01Z","title":"JSUT corpus: free large-scale Japanese speech corpus for end-to-end speech synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.00354","snapshot_observed_at":"2026-08-10T23:41:01.404222Z","title":"JSUT corpus: free large-scale Japanese speech corpus for end-to-end speech synthesis,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.404222Z"},"links":{"cited_paper":"/paper/1711.00354","citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:cbb93f21835cb26fb24e590fc557a2e4d30f1faee144538961843cdf8b22dd4d","observation_id":"b73612f4-d09b-4e05-8e46-eb109fdb70ce","resolution":{"observed_at":"2026-08-10T23:41:01.404222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.996286Z","title":"Multi-speaker TTS data,","venue":null,"work_id":"523d2b5b-8900-49a3-89e9-6dbf46ba77a3","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.413718Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:cda7f6870211a4043704d529e8bafae9b31f9a637813d868b9fcdbe28285444d","observation_id":"4a2e9241-8696-4630-ae8c-87cb57ecb2ae","resolution":{"observed_at":"2026-08-10T23:41:02.001771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.975565Z","title":"Phonemizer: Text to phones transcription for multiple languages in python,","venue":null,"work_id":"85ab3c2e-b714-49f5-a861-79b378e62673","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.419782Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:c7cf4c6eaf7dd30eec095bda719543e8c8d9f29c3934a5ba7a3fd8be3a85e243","observation_id":"3972753a-281c-48cd-8d64-58c05baed105","resolution":{"observed_at":"2026-08-10T23:41:01.981863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.948296Z","title":"NANSY++: Unified voice synthesis with neural analysis and synthesis,","venue":null,"work_id":"d74730e6-12b6-4bd5-8ae6-2772bf95721e","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.426365Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:38e1f94a6d67fb0de8d6bd04f1d141b422532b9a739e8635fb2957382e1568c2","observation_id":"83f93c23-0ea3-49da-bc02-620346aed726","resolution":{"observed_at":"2026-08-10T23:41:01.958515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.926232Z","title":"Language modeling with gated convolutional networks,","venue":null,"work_id":"c56454fe-7438-4ef8-80cd-8becf0012cd4","year":2017},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.439805Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:236d2b7e4525df5389e32054df1469bdfc5f3adaab7020074bad5e7f2ec695f7","observation_id":"eb81c13e-73ad-42df-bdce-8e2274db605a","resolution":{"observed_at":"2026-08-10T23:41:01.934226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.902030Z","title":"Fre-GAN: Adversarial frequency-consistent audio synthesis,","venue":null,"work_id":"67c2d6b7-c861-4d55-81fc-de55e22cf57f","year":2021},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.453287Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:4a3573d8a5d941fa9f07268b7449fa6ae15f21212bef37ee0263dce66f6df8a6","observation_id":"dd1b4f6f-e28d-4026-b08d-8bcceceaadab","resolution":{"observed_at":"2026-08-10T23:41:01.908739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.876447Z","title":"UTMOS: Utokyo-sarulab system for voicemos challenge 2022,","venue":null,"work_id":"c1f4ca08-c06c-4cf1-9f7f-ebc7fdbbe98d","year":2022},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.460239Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:ae4c3e1db80a9f31db20200a01c3f54285148d807d3aef8d65aa3d4a08c1c96a","observation_id":"966db735-0d84-4e0b-bec3-82e608b80322","resolution":{"observed_at":"2026-08-10T23:41:01.888958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.856170Z","title":"The blizzard challenge 2005: Evaluating corpus-based speech synthesis on common databases,","venue":null,"work_id":"918090fb-1a49-440e-8fd6-d2bfe5c4713e","year":2005},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.467343Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:9d63960823eab95b414e6a62f717d2e5ac030d2b07f62a8a7acd5a0c05e05e75","observation_id":"3704482b-c728-455c-ad3e-694cd1fe8531","resolution":{"observed_at":"2026-08-10T23:41:01.862661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.829083Z","title":"USAT: A universal speaker-adaptive text-to-speech approach,","venue":null,"work_id":"15d2067f-d228-40cb-80bb-eb184920aace","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.473731Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:c6c374f12c7385e76c7b446817ee91476ab964461f439002cab6daf31bf852bd","observation_id":"e17948f1-e53e-4637-a248-cd30562989d9","resolution":{"observed_at":"2026-08-10T23:41:01.841392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.795307Z","title":"Dual-branch modeling based on state-space model for speech enhancement,","venue":null,"work_id":"b8b94c46-d1cc-4e59-9fdd-f4139406d20f","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.484734Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:446c51016da784fbf3f37e5a9c2c3fb65b1f79d15e6211cad239ad4de6536ef6","observation_id":"d5973154-1774-4f7a-9d78-36add20b08ae","resolution":{"observed_at":"2026-08-10T23:41:01.803661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.764750Z","title":"V oicegrad: Non-parallel any-to-many voice conversion with annealed langevin dy- namics,","venue":null,"work_id":"2fddfa85-123f-4695-bc27-e692c5bdf94e","year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.490103Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:55250a966b1d771891feac3b93356fe495526c43f241e6acfaf8443f6120671a","observation_id":"5b5b390d-242c-4881-949c-f96c4c20a761","resolution":{"observed_at":"2026-08-10T23:41:01.779771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.724145Z","title":"Robust speech recognition via large-scale weak super- vision,","venue":null,"work_id":"67803c82-8477-4d6d-905e-1728562e3f4b","year":2023},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.495439Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:086dd14c50e1d786dff60aa7c7e1ce03dcd718b8b7da3e9937646f6303b6efe0","observation_id":"7f978c48-dc69-4df5-913d-dc6478aadf7d","resolution":{"observed_at":"2026-08-10T23:41:01.737629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T23:41:01.703417Z","title":"Visualizing data using t-SNE,","venue":null,"work_id":"0d623224-7300-415d-b796-00db2b05e0b4","year":2008},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.500859Z"},"links":{"citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:f1c67a84c939bd5a9c386148e07784024f6a35cc337a0dd503b003aeee938478","observation_id":"192c509f-3006-45f7-88c2-f350c5d28d0d","resolution":{"observed_at":"2026-08-10T23:41:01.710005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03926","last_updated":"2023-03-07T14:31:55Z","snapshot_observed_at":"2026-08-06T04:55:08.186019Z","submitted_at":"2023-03-07T14:31:55Z","title":"Speak Foreign Languages with Your Own Voice: Cross-Lingual Neural Codec Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.03926","snapshot_observed_at":"2026-08-10T23:41:01.505486Z","title":"Speak foreign languages with your own voice: Cross-lingual neural codec language modeling,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.505486Z"},"links":{"cited_paper":"/paper/2303.03926","citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:c1c6c4b90e7a7b328103b07bda761b5332364a5a87bde2cf3f86083fa7964d9d","observation_id":"96e5ca60-1ce3-47fe-b1c8-a37cf7fd8237","resolution":{"observed_at":"2026-08-10T23:41:01.505486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04904","last_updated":"2024-06-07T12:56:11Z","snapshot_observed_at":"2026-08-06T13:29:11.244845Z","submitted_at":"2024-06-07T12:56:11Z","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04904","snapshot_observed_at":"2026-08-10T23:41:01.512970Z","title":"XTTS: a massively multilingual zero-shot text-to-speech model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-10T23:41:01.512970Z"},"links":{"cited_paper":"/paper/2406.04904","citing_paper":"/paper/2412.20048"},"observation_digest":"sha256:72d771a26abdc4e4e5c2874b8af033a1b5c86bbbe9e5af33bdcefc92c1b1b06a","observation_id":"99f0a98a-20ff-4fe6-ac2b-0712939f2c70","resolution":{"observed_at":"2026-08-10T23:41:01.512970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.20048","last_updated":"2024-12-28T06:32:49Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-11T20:29:42.592115Z","submitted_at":"2024-12-28T06:32:49Z","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation"},"reference_resolution":{"displayed":75,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":5,"verified_exact":1,"verified_fuzzy":69},"total_outbound_references":75},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 75 of 75 outbound references and 0 inbound Pith citation observations for arXiv:2412.20048."}