{"as_of":"2026-08-22T21:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:15db795505604939c8ccaad70b5ad17b2eaa90491594c135ce34dc29712c1073","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":68,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":68,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:24:26.891680Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T18:15:21.399873Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-08-15T11:55:32.600679Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T12:26:37.300599Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2406.02430"},"observation_digest":"sha256:5c4f0d8a2717df384ba8cc12e8c9d2669df2a697b0518f40d8d10b187a4954f9","observation_id":"e3ced301-f751-4fa9-a4b5-f08d66b409ff","resolution":{"observed_at":"2026-05-15T12:26:37.432620Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-16T05:58:47.061997Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":134,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:77a8c33b1d1368209bd6ddb27f04dfcc3e72407a9c18a36ca7af294ba1aff6af","observation_id":"f9c429f0-e244-418f-b4c4-40b9260a9347","resolution":{"observed_at":"2026-05-16T06:06:41.498563Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-12T16:37:25.329513Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.329513Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:7b24d1d16664264a0482a039c404353cedb157426ed45d5d1fd25a36e70e27d5","observation_id":"7d176ddb-9b26-4ff6-935c-4a9d5e2baec5","resolution":{"observed_at":"2026-08-12T16:37:25.329513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-12T20:13:57.934932Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2411.13577","last_updated":"2024-11-26T09:20:48Z","snapshot_observed_at":"2026-08-20T13:09:54.981053Z","submitted_at":"2024-11-15T04:16:45Z","title":"WavChat: A Survey of Spoken Dialogue Models","version":2},"reference_index":177,"source":"pdf_text","source_observed_at":"2026-08-12T20:13:57.934932Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2411.13577"},"observation_digest":"sha256:9925861672eb0c4041c84a104ca87d93c54104c656feb388117c364d1804435c","observation_id":"aab0afda-e84e-46d6-8d38-4f2dba8dd974","resolution":{"observed_at":"2026-08-12T20:13:57.934932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T22:51:49.895990Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.03074","last_updated":"2024-12-04T06:52:03Z","snapshot_observed_at":"2026-08-16T20:33:32.352473Z","submitted_at":"2024-12-04T06:52:03Z","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T22:51:49.895990Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.03074"},"observation_digest":"sha256:7218b8c7e0424575e1674f682e957547ec33211e4c5745d3bcfeac232916ecde","observation_id":"ff9c97c0-aaa2-497e-b7b9-dd602b6bb6bc","resolution":{"observed_at":"2026-08-11T22:51:49.895990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T18:16:26.922799Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.08112","last_updated":"2024-12-11T05:39:12Z","snapshot_observed_at":"2026-08-16T05:34:19.657013Z","submitted_at":"2024-12-11T05:39:12Z","title":"Aligner-Guided Training Paradigm: Advancing Text-to-Speech Models with Aligner Guided Duration","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T18:16:26.922799Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.08112"},"observation_digest":"sha256:089e91c729958e9a6ceb0190be655e04ba49580e850517c7fe6807d94211d061","observation_id":"2ec1ae0d-b63b-44fb-a6c7-f2372c92b755","resolution":{"observed_at":"2026-08-11T18:16:26.922799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T18:16:12.922242Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.08117","last_updated":"2024-12-11T05:55:06Z","snapshot_observed_at":"2026-08-12T04:41:01.550373Z","submitted_at":"2024-12-11T05:55:06Z","title":"LatentSpeech: Latent Diffusion for Text-To-Speech Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T18:16:12.922242Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.08117"},"observation_digest":"sha256:d6c42d334f484dacba37aeaf94720f25d6f62cc0d20adc57e2cda035b87017b3","observation_id":"46ba7a1e-1aaf-4d62-9f4f-a52768b8babd","resolution":{"observed_at":"2026-08-11T18:16:12.922242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T17:17:48.944939Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.09168","last_updated":"2024-12-12T10:55:57Z","snapshot_observed_at":"2026-08-19T16:13:49.463844Z","submitted_at":"2024-12-12T10:55:57Z","title":"YingSound: Video-Guided Sound Effects Generation with Multi-modal Chain-of-Thought Controls","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T17:17:48.944939Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.09168"},"observation_digest":"sha256:f4ef1aec891f41aa6372f72a4183ec6863ad70fceb1ef812e458983b3b8aaa5a","observation_id":"4c17a08f-eb8d-4e28-a643-276f24aa57c7","resolution":{"observed_at":"2026-08-11T17:17:48.944939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T14:05:25.870697Z","title":"FastSpeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.12512","last_updated":"2024-12-17T04:06:53Z","snapshot_observed_at":"2026-08-12T06:01:15.050371Z","submitted_at":"2024-12-17T04:06:53Z","title":"Libri2Vox Dataset: Target Speaker Extraction with Diverse Speaker Conditions and Synthetic Data","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T14:05:25.870697Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.12512"},"observation_digest":"sha256:e6c63460b29884e25a1de303fd04d61866d2a5ae5fb845bbfcaaa4bb30d00adc","observation_id":"2ebe90da-d97e-4d12-a128-be2ba4309274","resolution":{"observed_at":"2026-08-11T14:05:25.870697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-11T04:34:07.851676Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2412.18748","last_updated":"2024-12-31T07:27:26Z","snapshot_observed_at":"2026-08-22T18:19:42.428330Z","submitted_at":"2024-12-25T02:41:13Z","title":"Towards Expressive Video Dubbing with Multiscale Multimodal Context Interaction","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T04:34:07.851676Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2412.18748"},"observation_digest":"sha256:fa9f71152bc873da8bc73b987df2037314956ab749f9f7bb7cd0bb82ab44ef3e","observation_id":"8bd7cb49-60bb-4b7c-af54-203db6ed751b","resolution":{"observed_at":"2026-08-11T04:34:07.851676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-10T22:41:36.264122Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.03181","last_updated":"2025-04-15T19:16:19Z","snapshot_observed_at":"2026-08-17T00:51:35.940686Z","submitted_at":"2025-01-02T02:00:15Z","title":"FaceSpeak: Expressive and High-Quality Speech Synthesis from Human Portraits of Different Styles","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T22:41:36.264122Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2501.03181"},"observation_digest":"sha256:cfbd846a441c82ce53cdf260ce8765651d4c4b089c2db8f1b0c88a6c229da288","observation_id":"68c217f8-b477-4d00-a6f6-1b230cea27e9","resolution":{"observed_at":"2026-08-10T22:41:36.264122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-10T21:09:34.949717Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.06276","last_updated":"2025-01-10T12:10:30Z","snapshot_observed_at":"2026-08-13T15:16:15.139371Z","submitted_at":"2025-01-10T12:10:30Z","title":"PROEMO: Prompt-Driven Text-to-Speech Synthesis Based on Emotion and Intensity Control","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T21:09:34.949717Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2501.06276"},"observation_digest":"sha256:5fdd0af179a1562ab1743790acd6b47e781f8fdccafd981fde5483acffc0a43d","observation_id":"304b3469-c214-49b8-b21e-bf8811bbf1c3","resolution":{"observed_at":"2026-08-10T21:09:34.949717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-10T20:15:28.530654Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.09104","last_updated":"2025-01-20T18:35:40Z","snapshot_observed_at":"2026-08-18T01:52:36.466119Z","submitted_at":"2025-01-15T19:42:41Z","title":"A Non-autoregressive Model for Joint STT and TTS","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T20:15:28.530654Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2501.09104"},"observation_digest":"sha256:2b40717d60f0561c78f8d0e55c287ae4716d6c6be574cdaad84fca82ed04d381","observation_id":"25f1d255-6d72-4087-87ff-f4a4ff22e5fa","resolution":{"observed_at":"2026-08-10T20:15:28.530654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-09T20:50:17.265052Z","title":"FastSpeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-21T15:03:52.661179Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.265052Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:d79c81468a2d5cf8cbc2d42cf584c23fe82b1ea79c77329f90417644b99ec151","observation_id":"b3401b66-54ee-4800-a4b7-f3fed282a958","resolution":{"observed_at":"2026-08-09T20:50:17.265052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-09T16:43:08.853890Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2502.01084","last_updated":"2025-02-13T09:25:03Z","snapshot_observed_at":"2026-08-18T03:00:24.814227Z","submitted_at":"2025-02-03T05:53:59Z","title":"Continuous Autoregressive Modeling with Stochastic Monotonic Alignment for Speech Synthesis","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-09T16:43:08.853890Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2502.01084"},"observation_digest":"sha256:85860326a458e2dd031238030d648d8581da8b6114d18d57520a7fcdf3ba3b54","observation_id":"30595c40-9843-41ae-99a8-e0c97676db0f","resolution":{"observed_at":"2026-08-09T16:43:08.853890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-09T05:54:18.722514Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2502.03128","last_updated":"2025-02-05T12:36:21Z","snapshot_observed_at":"2026-08-18T17:13:14.270187Z","submitted_at":"2025-02-05T12:36:21Z","title":"Metis: A Foundation Speech Generation Model with Masked Generative Pre-training","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-09T05:54:18.722514Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2502.03128"},"observation_digest":"sha256:783508d3aa426d998c27b9fcc74c0fafaea66b7fe03e3cf91e857e25c930360f","observation_id":"cfe4fe30-e4b7-4e03-be0a-e17f861970d9","resolution":{"observed_at":"2026-08-09T05:54:18.722514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T18:34:12.985216Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2502.10329","last_updated":"2025-02-14T17:43:01Z","snapshot_observed_at":"2026-08-12T13:34:08.056062Z","submitted_at":"2025-02-14T17:43:01Z","title":"VocalCrypt: Novel Active Defense Against Deepfake Voice Based on Masking Effect","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T18:34:12.985216Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2502.10329"},"observation_digest":"sha256:08bcea3a5df56b6c9630e4047a4085ccca25d76a71473055b3fec909b65ac4e2","observation_id":"c19b05be-bb84-4943-a068-019d9501fb15","resolution":{"observed_at":"2026-08-07T18:34:12.985216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-16T11:24:26.891680Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2504.15663","last_updated":"2025-04-22T07:40:35Z","snapshot_observed_at":"2026-08-21T13:58:25.785820Z","submitted_at":"2025-04-22T07:40:35Z","title":"FADEL: Uncertainty-aware Fake Audio Detection with Evidential Deep Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T11:24:26.891680Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2504.15663"},"observation_digest":"sha256:2743abbe954e75236158ca25e1a79a71f9da37e763f671e8f47548a6acddd753","observation_id":"275a9f07-8958-4fb7-8499-07f61f28cec4","resolution":{"observed_at":"2026-08-16T11:24:26.891680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2505.00409","last_updated":"2026-05-17T17:39:35Z","snapshot_observed_at":"2026-08-16T23:05:41.788895Z","submitted_at":"2025-05-01T09:03:03Z","title":"Perceptual implications of automatic anonymization in pathological speech","version":3},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-22T17:58:15.380780Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.00409"},"observation_digest":"sha256:0d6d6ac342b7053b651df42ef4192bb9a31af7fb7d5550084e28e57648dc295f","observation_id":"4c31c211-b8d0-40f6-8ce2-44ce49450fd3","resolution":{"observed_at":"2026-05-22T18:01:54.538397Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-15T23:22:41.105359Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.04885","last_updated":"2025-05-08T01:48:06Z","snapshot_observed_at":"2026-08-19T15:04:41.579527Z","submitted_at":"2025-05-08T01:48:06Z","title":"A Multi-Agent AI Framework for Immersive Audiobook Production through Spatial Audio and Neural Narration","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T23:22:41.105359Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.04885"},"observation_digest":"sha256:52dff6a7ab83d07dd0980025d48868cd10af47ee84e0581b60d01e959df2b8e1","observation_id":"ea6bd5fc-2563-43c4-949b-1e27e10d4eeb","resolution":{"observed_at":"2026-08-15T23:22:41.105359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-15T20:15:05.415737Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.13771","last_updated":"2025-05-19T23:12:25Z","snapshot_observed_at":"2026-08-17T15:41:10.869818Z","submitted_at":"2025-05-19T23:12:25Z","title":"Score-Based Training for Energy-Based TTS Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T20:15:05.415737Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.13771"},"observation_digest":"sha256:f5af3ff996fe6710767984ef45b2fe5748ed3f8de2f954cf89a0707c97e350fa","observation_id":"e6c22446-75de-408e-9bdf-9c1dddaee454","resolution":{"observed_at":"2026-08-15T20:15:05.415737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T14:52:16.526730Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.17410","last_updated":"2025-05-23T02:54:52Z","snapshot_observed_at":"2026-08-14T07:05:12.428820Z","submitted_at":"2025-05-23T02:54:52Z","title":"LLM-based Generative Error Correction for Rare Words with Synthetic Data and Phonetic Context","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:52:16.526730Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.17410"},"observation_digest":"sha256:eb7f4e7c520f147293efe7434cc0041acdbb9a5e3e8cab4f41e597b28281a0c7","observation_id":"b7501c1f-dd39-4816-9344-aa9aacd8daad","resolution":{"observed_at":"2026-08-07T14:52:16.526730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T14:33:38.162651Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.18453","last_updated":"2025-05-24T01:26:02Z","snapshot_observed_at":"2026-08-16T12:47:56.342175Z","submitted_at":"2025-05-24T01:26:02Z","title":"MPE-TTS: Customized Emotion Zero-Shot Text-To-Speech Using Multi-Modal Prompt","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:38.162651Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.18453"},"observation_digest":"sha256:5e490257ef83a177209880065d6a763406f88702aa9956c84c7ceb4e3bee73dd","observation_id":"d013a140-7618-4028-8f5d-28a9f72203de","resolution":{"observed_at":"2026-08-07T14:33:38.162651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T14:25:01.464367Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.19119","last_updated":"2025-05-25T12:22:00Z","snapshot_observed_at":"2026-08-09T08:08:38.091653Z","submitted_at":"2025-05-25T12:22:00Z","title":"CloneShield: A Framework for Universal Perturbation Against Zero-Shot Voice Cloning","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T14:25:01.464367Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.19119"},"observation_digest":"sha256:dfd3a8d657a82b9b42d426d8c20f7276c6e0937a9762d3b69f052a84867a78c8","observation_id":"bd0474c8-cad5-43b2-acc6-e2e05a9ca0d3","resolution":{"observed_at":"2026-08-07T14:25:01.464367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T14:22:31.169600Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-17T20:29:32.398121Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.169600Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:a05de59992bda86774ee611615b625dd62c05674ff7c17d45fd46d9f8669e2a0","observation_id":"7fc88f8e-e561-4c13-89a6-35e5a654f82f","resolution":{"observed_at":"2026-08-07T14:22:31.169600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T13:21:14.783720Z","title":"Available: https://arxiv.org/abs/2006.04558","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.22024","last_updated":"2025-05-28T06:46:13Z","snapshot_observed_at":"2026-08-15T09:33:30.498434Z","submitted_at":"2025-05-28T06:46:13Z","title":"RESOUND: Speech Reconstruction from Silent Videos via Acoustic-Semantic Decomposed Modeling","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T13:21:14.783720Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.22024"},"observation_digest":"sha256:801947b2e57f27dacf3b8ac571fe3f2dd852607b95cdcf4d7c0dc1adcd8cfabe","observation_id":"bf85582c-7e86-4658-a5b5-f5e2cdf7c42d","resolution":{"observed_at":"2026-08-07T13:21:14.783720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T12:00:10.317617Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.00832","last_updated":"2025-06-01T04:33:37Z","snapshot_observed_at":"2026-08-20T11:37:13.288169Z","submitted_at":"2025-06-01T04:33:37Z","title":"Counterfactual Activation Editing for Post-hoc Prosody and Mispronunciation Correction in TTS Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:00:10.317617Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2506.00832"},"observation_digest":"sha256:dd08c7499093fc9e340aa1044ccecca75e1816f8612902f0a4ccdc7cb09ec43e","observation_id":"fc3de9b6-9a65-4842-a3ef-4433dcddfc69","resolution":{"observed_at":"2026-08-07T12:00:10.317617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T04:45:19.164512Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.09874","last_updated":"2025-07-10T19:47:47Z","snapshot_observed_at":"2026-08-17T02:21:03.433032Z","submitted_at":"2025-06-11T15:43:08Z","title":"UmbraTTS: Adapting Text-to-Speech to Environmental Contexts with Flow Matching","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:19.164512Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2506.09874"},"observation_digest":"sha256:3003480756a8169b34063122a43799ad804851e143a712f256a4aebb4c95026e","observation_id":"e0cd1bd0-62d1-4a9c-8e8c-c581427ab987","resolution":{"observed_at":"2026-08-07T04:45:19.164512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T23:53:42.127548Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.16020","last_updated":"2025-06-19T04:34:53Z","snapshot_observed_at":"2026-08-17T04:20:17.436262Z","submitted_at":"2025-06-19T04:34:53Z","title":"VS-Singer: Vision-Guided Stereo Singing Voice Synthesis with Consistency Schr\\\"odinger Bridge","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T23:53:42.127548Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2506.16020"},"observation_digest":"sha256:dbf42443b0f32370bcf9c090b0eaba92d1471ae7cd1a42c792b2f854212596a9","observation_id":"762799b2-c25c-48d3-be60-9f847c2a8a4d","resolution":{"observed_at":"2026-08-06T23:53:42.127548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-15T19:32:30.944300Z","title":"InICASSP 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 6493–6497","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16381","last_updated":"2025-06-19T15:08:01Z","snapshot_observed_at":"2026-08-16T17:50:48.232882Z","submitted_at":"2025-06-19T15:08:01Z","title":"InstructTTSEval: Benchmarking Complex Natural-Language Instruction Following in Text-to-Speech Systems","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-15T19:32:30.944300Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2506.16381"},"observation_digest":"sha256:938100b01408b2d2910da5e8fdbc75434eab036dbd6dd9b132d698fbc23e6eae","observation_id":"d5e5367b-6e78-4ca4-b051-14f110120e45","resolution":{"observed_at":"2026-08-15T19:32:30.944300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T22:29:28.366767Z","title":null,"venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.21478","last_updated":"2025-06-26T17:07:45Z","snapshot_observed_at":"2026-08-10T02:02:49.811618Z","submitted_at":"2025-06-26T17:07:45Z","title":"SmoothSinger: A Conditional Diffusion Model for Singing Voice Synthesis with Multi-Resolution Architecture","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T22:29:28.366767Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2506.21478"},"observation_digest":"sha256:c1fe627f6983afbbef83b87b67ccfcf7ff8cadeb17286fd97f19d3dc93636e81","observation_id":"51258cdc-1a04-47bc-b851-77e66cfb4e41","resolution":{"observed_at":"2026-08-06T22:29:28.366767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2506.23552","last_updated":"2026-05-15T04:47:28Z","snapshot_observed_at":"2026-08-16T21:02:47.373194Z","submitted_at":"2025-06-30T06:51:40Z","title":"JAM-Flow: Joint Audio-Motion Synthesis with Flow Matching","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-22T00:46:39.196042Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2506.23552"},"observation_digest":"sha256:7692cfe6fa9bcc13f370d37cb9d68491f35fa9a9fc0c667d2740123bdb5b0ae8","observation_id":"b5600eec-68cb-4ea2-b308-f9cfa7d32d1f","resolution":{"observed_at":"2026-05-22T00:50:51.001574Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T21:23:47.831204Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.00324","last_updated":"2025-06-30T23:41:04Z","snapshot_observed_at":"2026-08-20T18:09:28.126757Z","submitted_at":"2025-06-30T23:41:04Z","title":"Collecting, Curating, and Annotating Good Quality Speech deepfake dataset for Famous Figures: Process and Challenges","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:23:47.831204Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.00324"},"observation_digest":"sha256:aa41e50c49565f590bb891747cbaa165ada1865dfff138fae6ab5444d7170d0f","observation_id":"8341b7aa-fa69-4608-9424-d0d851b6f797","resolution":{"observed_at":"2026-08-06T21:23:47.831204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T20:01:35.420849Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.08012","last_updated":"2025-07-05T10:59:00Z","snapshot_observed_at":"2026-08-18T22:50:27.977401Z","submitted_at":"2025-07-05T10:59:00Z","title":"RepeaTTS: Towards Feature Discovery through Repeated Fine-Tuning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:01:35.420849Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.08012"},"observation_digest":"sha256:aa5c798d4ff2d1e74db05970a9d1125fd4861c78ae56a8a3b154e4db20d2ce2c","observation_id":"5402b3f9-b82d-4aa5-aab3-8a802652788f","resolution":{"observed_at":"2026-08-06T20:01:35.420849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T16:55:50.287364Z","title":"Xinsheng Wang, Mingqi Jiang, Ziyang Ma, Ziyu Zhang, Songxiang Liu, Linqin Li, Zheng Liang, Qixi Zheng, Rui Wang, Xiaoqin Feng, et al","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.12197","last_updated":"2025-07-16T12:47:09Z","snapshot_observed_at":"2026-08-07T14:59:35.759272Z","submitted_at":"2025-07-16T12:47:09Z","title":"Quantize More, Lose Less: Autoregressive Generation from Residually Quantized Speech Representations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T16:55:50.287364Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.12197"},"observation_digest":"sha256:e05d0772962062789c438bb8be61371fec1772f83cc6bcd355e7517a4ea0b3f9","observation_id":"628de9bd-5f27-43bd-ab8d-750b22b24240","resolution":{"observed_at":"2026-08-06T16:55:50.287364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-08-13T23:38:52.637908Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:9854c107e92db19ccc236a8bf35e75818156dddb56dd357208317ef7232abc79","observation_id":"133feca1-a02c-4ef2-a45a-904b8aa3e946","resolution":{"observed_at":"2026-05-16T05:59:51.194912Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T15:14:30.485844Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.16875","last_updated":"2025-07-22T09:38:30Z","snapshot_observed_at":"2026-08-21T14:11:10.183676Z","submitted_at":"2025-07-22T09:38:30Z","title":"Technical report: Impact of Duration Prediction on Speaker-specific TTS for Indian Languages","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T15:14:30.485844Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.16875"},"observation_digest":"sha256:853ffe1fd1ae8c1ee473da2fbe68fdfb27c1b9e07a843eaeef3061a5c551ff79","observation_id":"dbec8d13-d99a-4ef8-bbec-77b4844aaab6","resolution":{"observed_at":"2026-08-06T15:14:30.485844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T13:54:40.218211Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.19894","last_updated":"2026-08-15T13:49:25Z","snapshot_observed_at":"2026-08-20T23:09:07.959533Z","submitted_at":"2025-07-26T09:49:57Z","title":"Generative Model Unlearning: A Survey through Target Events, Unlearning Operators, and Evaluation Protocols","version":1},"reference_index":185,"source":"pdf_text","source_observed_at":"2026-08-06T13:54:40.218211Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.19894"},"observation_digest":"sha256:de534b58c79cd34f04f2538175040f0b18cb6db3f96d9827a3c33a699cb876a3","observation_id":"2b9a4059-25c8-49d5-9df7-ee5e831806cc","resolution":{"observed_at":"2026-08-06T13:54:40.218211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T11:34:08.523005Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.22612","last_updated":"2025-08-29T06:09:29Z","snapshot_observed_at":"2026-08-14T20:29:00.417184Z","submitted_at":"2025-07-30T12:31:11Z","title":"Adaptive Duration Model for Text Speech Alignment","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T11:34:08.523005Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2507.22612"},"observation_digest":"sha256:6194b2a09159592695ebcbddd33fefee4b57f385faafa6e99db4a2b2b50472ad","observation_id":"10183caa-b808-4807-92b6-767070d9fb8e","resolution":{"observed_at":"2026-08-06T11:34:08.523005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T05:05:27.864867Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech.arXiv preprint arXiv:2006.04558, 2020","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2508.02391","last_updated":"2025-08-04T13:17:49Z","snapshot_observed_at":"2026-08-08T02:50:04.116531Z","submitted_at":"2025-08-04T13:17:49Z","title":"Inference-time Scaling for Diffusion-based Audio Super-resolution","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T05:05:27.864867Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2508.02391"},"observation_digest":"sha256:f20e4fd6633aa9111270c88e822977efe0a3e37b7e23af374aeeef431bb9a2cc","observation_id":"68004d0d-dddb-45cc-b8d5-53f7c625913f","resolution":{"observed_at":"2026-08-06T05:05:27.864867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-05T22:17:38.247156Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2508.07302","last_updated":"2025-08-12T02:31:09Z","snapshot_observed_at":"2026-08-21T18:56:28.717588Z","submitted_at":"2025-08-10T11:31:13Z","title":"XEmoRAG: Cross-Lingual Emotion Transfer with Controllable Intensity Using Retrieval-Augmented Generation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T22:17:38.247156Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2508.07302"},"observation_digest":"sha256:6f8b413b6cf75c3dcf3ec8b2bdde2f77c013b1eb95157483bde7a42197c7a748","observation_id":"747a176c-8695-4ef4-a090-5eede55c3179","resolution":{"observed_at":"2026-08-05T22:17:38.247156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-05T16:31:05.578155Z","title":"Renqian Luo, Xu Tan, Rui Wang, Tao Qin, Jinzhu Li, Sheng Zhao, Enhong Chen, and Tie-Yan Liu","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2508.18440","last_updated":"2025-08-25T19:39:20Z","snapshot_observed_at":"2026-08-18T10:26:22.149461Z","submitted_at":"2025-08-25T19:39:20Z","title":"SwiftF0: Fast and Accurate Monophonic Pitch Detection","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T16:31:05.578155Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2508.18440"},"observation_digest":"sha256:2c7f9d4c741e811caa30c14b37d9063eab6deeecfbb0355241ee53c50d7ae032","observation_id":"7d0148a8-a446-4da9-8b35-6f0dd695f5e8","resolution":{"observed_at":"2026-08-05T16:31:05.578155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-05T12:40:10.529671Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2509.01391","last_updated":"2025-09-01T11:36:37Z","snapshot_observed_at":"2026-08-16T02:26:20.533734Z","submitted_at":"2025-09-01T11:36:37Z","title":"MixedG2P-T5: G2P-free Speech Synthesis for Mixed-script texts using Speech Self-Supervised Learning and Language Model","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T12:40:10.529671Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2509.01391"},"observation_digest":"sha256:b25e40581daa723b7d14ac6a9f168fb31ff161ba1d0b42f9c5341ae5848dc5c9","observation_id":"1257031f-77b7-42fc-a4b6-f15c4431988c","resolution":{"observed_at":"2026-08-05T12:40:10.529671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2509.20086","last_updated":"2026-05-08T13:18:21Z","snapshot_observed_at":"2026-08-14T17:01:43.724752Z","submitted_at":"2025-09-24T13:05:09Z","title":"OLaPh: Optimal Language Phonemizer","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-18T14:24:51.262622Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2509.20086"},"observation_digest":"sha256:c5189c696c112d9d07234c1ff83046f36cd3a477b3a3784099c23c881e4a22a9","observation_id":"e9e67638-e956-49a0-bc13-2c90fd26f737","resolution":{"observed_at":"2026-05-18T14:26:28.192861Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-15T15:51:37.338770Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2509.21597","last_updated":"2026-06-03T17:37:13Z","snapshot_observed_at":"2026-08-18T03:36:37.563547Z","submitted_at":"2025-09-25T21:09:40Z","title":"AUDDT: A Unified Benchmark Toolkit for Audio and Speech Deepfake Detectors","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T15:51:37.338770Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2509.21597"},"observation_digest":"sha256:21bf4ec934c1413a1b50ba10ce8094655de51b438e5573b60cf7a426e18ba000","observation_id":"80ba89bc-1bef-46a9-97a6-e5657b0b5713","resolution":{"observed_at":"2026-08-15T15:51:37.338770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-04T00:27:32.435469Z","title":"Fastspeech 2: Fast and high-quality end-to- end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2511.01056","last_updated":"2026-07-20T02:49:17Z","snapshot_observed_at":"2026-08-21T13:51:21.591167Z","submitted_at":"2025-11-02T19:18:38Z","title":"WhisperVC: Decoupled Cross-Domain Alignment and Speech Generation for Low-Resource Whisper-to-Normal Conversion","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T00:27:32.435469Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2511.01056"},"observation_digest":"sha256:d1e9f6d58193ecaf14fa50786d048387f2db6675ff6a983ac00cf38c028f085e","observation_id":"7dac97be-17ba-447e-ad96-257d4e781f18","resolution":{"observed_at":"2026-08-04T00:27:32.435469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2511.20657","last_updated":"2026-05-02T12:09:38Z","snapshot_observed_at":"2026-08-13T15:15:19.905394Z","submitted_at":"2025-10-11T07:40:36Z","title":"Intelligent Agents with Emotional Intelligence: Current Trends, Challenges, and Future Prospects","version":2},"reference_index":140,"source":"pdf_text","source_observed_at":"2026-05-18T08:11:31.181704Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2511.20657"},"observation_digest":"sha256:05977c5e181a964e9d8a655613251622047bcd6a43a4e8b95da59484cda7fbe8","observation_id":"e2bb892b-6864-48fd-8c99-9ea5d662fef8","resolution":{"observed_at":"2026-05-18T08:12:30.232674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-03T00:11:16.003779Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech.arXiv preprint arXiv:2006.04558,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2602.12304","last_updated":"2026-07-23T11:24:08Z","snapshot_observed_at":"2026-08-19T16:13:28.473895Z","submitted_at":"2026-02-12T03:25:41Z","title":"OmniCustom: Sync Audio-Video Customization Via Joint Audio-Video Generation Model","version":5},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-03T00:11:16.003779Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2602.12304"},"observation_digest":"sha256:10a82aa97953acf67d0fe0ddfcc0a89293a641a6d9c7d455588312775ffe3612","observation_id":"795dc8f7-e99f-4731-a327-18aba2d1db48","resolution":{"observed_at":"2026-08-03T00:11:16.003779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2602.22029","last_updated":"2026-05-05T13:00:43Z","snapshot_observed_at":"2026-08-15T02:10:54.663927Z","submitted_at":"2026-02-24T06:43:27Z","title":"MIDI-Informed Singing Accompaniment Generation in a Compositional Song Pipeline","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-15T20:17:45.920234Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2602.22029"},"observation_digest":"sha256:d2cf12a7d44bd24078bd84192976fc167cbd36c8a8699ff583b311dde9b20618","observation_id":"759b864d-0c3b-408b-9987-b89f0e364d76","resolution":{"observed_at":"2026-05-15T20:20:18.020832Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-02T18:35:56.164869Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2603.08173","last_updated":"2026-07-18T12:50:12Z","snapshot_observed_at":"2026-08-16T20:29:56.796050Z","submitted_at":"2026-03-09T09:53:05Z","title":"Evolution Strategy-Based Calibration for Low-Bit Quantization of Speech Models","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T18:35:56.164869Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2603.08173"},"observation_digest":"sha256:43802a75d6695dd199f3f5e289e2ccf35eb1535ead1927ec4f2e3fef32e74f62","observation_id":"8d0d5933-a97a-415d-99c4-b9cfd8414a70","resolution":{"observed_at":"2026-08-02T18:35:56.164869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2604.16056","last_updated":"2026-08-03T16:51:20Z","snapshot_observed_at":"2026-08-16T12:42:56.856916Z","submitted_at":"2026-04-17T13:30:59Z","title":"AST: Adaptive, Seamless, and Training-Free Precise Speech Editing","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T07:59:34.622096Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2604.16056"},"observation_digest":"sha256:dcdf46590ba21992ef0f60dc881c7ce4f18e926ff191d64e624b547427086965","observation_id":"6c288869-8085-4a6a-8da6-d7d552cceac1","resolution":{"observed_at":"2026-05-10T08:02:25.252636Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-04T05:29:35.464288Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech.arXiv preprint arXiv:2006.04558, 2020","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2604.16056","last_updated":"2026-08-03T16:51:20Z","snapshot_observed_at":"2026-08-16T12:42:56.856916Z","submitted_at":"2026-04-17T13:30:59Z","title":"AST: Adaptive, Seamless, and Training-Free Precise Speech Editing","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T05:29:35.464288Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2604.16056"},"observation_digest":"sha256:62e9ef0375bd615c8ef578e9fc58d57c2a7552a914ce475a5635c9ed59ab9c42","observation_id":"9bf430db-384e-425a-a788-69cf379c3eb3","resolution":{"observed_at":"2026-08-04T05:29:35.464288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2605.00861","last_updated":"2026-04-21T10:34:41Z","snapshot_observed_at":"2026-07-06T23:14:11.211164Z","submitted_at":"2026-04-21T10:34:41Z","title":"Voice Mapping of Text-to-Speech Systems: A Metric-Based Approach for Voice Quality Assessment","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T01:07:50.243903Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2605.00861"},"observation_digest":"sha256:5e7f623745214e250914d9567bd5ae762096b2958f5d4d6921a6dc03644baea2","observation_id":"8c7c3b5d-04c4-40d9-9054-e0883607fd77","resolution":{"observed_at":"2026-05-10T01:10:09.304537Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2605.01515","last_updated":"2026-05-02T16:07:12Z","snapshot_observed_at":"2026-08-14T18:03:08.715190Z","submitted_at":"2026-05-02T16:07:12Z","title":"MelShield: Robust Mel-Domain Audio Watermarking for Provenance Attribution of AI Generated Synthesized Speech","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:05:58.034000Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2605.01515"},"observation_digest":"sha256:37633d46bc979d13b5e82376991a8e89cd637b8cafa1abdc0095581289527d2f","observation_id":"7d3b91d2-5017-4284-a5a8-6182bb24e22d","resolution":{"observed_at":"2026-05-11T09:20:59.810611Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2605.09971","last_updated":"2026-05-11T04:26:51Z","snapshot_observed_at":"2026-07-06T23:21:59.096464Z","submitted_at":"2026-05-11T04:26:51Z","title":"HapticLDM: A Diffusion Model for Text-to-Vibrotactile Generation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-12T04:11:29.782569Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2605.09971"},"observation_digest":"sha256:6b3ca7b0f1d37a2f04c11c5473a8d3682ceae39ccde4191edf5ab3d1af47eab7","observation_id":"82fed65f-3184-4ec3-8c2e-3ca5adf453b1","resolution":{"observed_at":"2026-05-12T06:31:28.158629Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2606.09295","last_updated":"2026-06-08T10:03:11Z","snapshot_observed_at":"2026-08-16T20:40:48.319555Z","submitted_at":"2026-06-08T10:03:11Z","title":"N\\\"ushuVoice: Reviving the Voice of Endangered N\\\"ushu with Pitch-Aware Text-to-Speech","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-27T16:39:49.263497Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2606.09295"},"observation_digest":"sha256:ab237fb971244e628acb7b2ee38c1e3748934f3ae7d83d60a261df87ae2b204b","observation_id":"d04f7cec-2598-4d24-b07d-362ee582d878","resolution":{"observed_at":"2026-07-03T01:17:30.978541Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2606.19209","last_updated":"2026-06-18T16:42:05Z","snapshot_observed_at":"2026-08-13T15:14:39.992884Z","submitted_at":"2026-06-17T15:45:43Z","title":"FineCombo-TTS: Collaborative and Precise Controllable Speech Synthesis Using Text Descriptions and Reference Speech","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T19:09:33.605232Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2606.19209"},"observation_digest":"sha256:fd484e9769a3b1e4fcb80bdc55aadc5549ae15baca781bfbe9a0d3c0ee5e0fdf","observation_id":"abb0c836-e6d8-4c3e-b29c-9ba1b2ef9c9f","resolution":{"observed_at":"2026-07-04T02:49:24.574696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2606.21457","last_updated":"2026-06-19T14:13:16Z","snapshot_observed_at":"2026-08-21T13:07:09.820607Z","submitted_at":"2026-06-19T14:13:16Z","title":"DisSpeech: Low-Resource Controllable Mandarin Stuttered Speech Synthesis for ASR Augmentation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-26T12:53:34.428544Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2606.21457"},"observation_digest":"sha256:200c0a154f1d07ecad96f10ad90de8d36af61d37bdba115f9f8adbb39e5d41e7","observation_id":"8a76256e-c556-48ec-b6c9-e65bce9e280c","resolution":{"observed_at":"2026-07-04T07:49:38.611490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2606.22310","last_updated":"2026-06-21T02:35:22Z","snapshot_observed_at":"2026-08-16T06:14:22.260530Z","submitted_at":"2026-06-21T02:35:22Z","title":"Learning to Evade: Adaptive Attacks on Audio Watermarking","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-26T10:10:18.351233Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2606.22310"},"observation_digest":"sha256:981ab8c54dcdf47894694c1c12844e93a00d6cb66e17e025853b35dec65dad3e","observation_id":"f4de4787-a61c-49b9-8988-98a03e38c516","resolution":{"observed_at":"2026-07-04T09:19:43.962125Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2606.30580","last_updated":"2026-06-29T17:22:27Z","snapshot_observed_at":"2026-08-12T15:40:30.941373Z","submitted_at":"2026-06-29T17:22:27Z","title":"MeloDISinger: Melody-Aware & Duration-Preserving Singing Voice Editing with Audio Infilling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-30T03:20:05.125635Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2606.30580"},"observation_digest":"sha256:b17150add17aeb2e1f33add5c718c47f3b191d7b575991cbaddbf0a85a3175f3","observation_id":"6f0c767e-c6a4-4b61-86fb-329d8473eeb5","resolution":{"observed_at":"2026-06-30T03:24:12.484621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2606.31247","last_updated":"2026-06-30T07:24:10Z","snapshot_observed_at":"2026-08-13T09:35:09.713263Z","submitted_at":"2026-06-30T07:24:10Z","title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","version":1},"reference_index":160,"source":"arxiv_source","source_observed_at":"2026-07-01T03:50:26.873406Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2606.31247"},"observation_digest":"sha256:81bfdaf6fc96ac056d61d4374d1075be5315dbd26978b889f78e58b77b65ebf1","observation_id":"401d0283-227c-44fd-924b-d3115d05a1fd","resolution":{"observed_at":"2026-07-01T11:45:47.203091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":"2006.04558","doi":null,"metadata_source":"pith","pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-08T18:15:21.399873Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","venue":"eess.AS","work_id":"415023d2-900f-4bda-a774-38e2d95b7a8f","year":2020},"citing_paper":{"arxiv_id":"2607.06054","last_updated":"2026-07-07T09:31:19Z","snapshot_observed_at":"2026-08-14T10:10:54.178445Z","submitted_at":"2026-07-07T09:31:19Z","title":"BlueMagpie-TTS: A Token-Efficient Tokenizer, Language Model, and TTS for Taiwanese-Accent Code-Switching Speech","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-08T18:09:22.207379Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2607.06054"},"observation_digest":"sha256:737055120f2b4fe72de3bbb1696f1fe82553805d199218addf9969fd3bca8273","observation_id":"6fa43e09-9a85-4bbf-9982-bcb5baabf37d","resolution":{"observed_at":"2026-07-08T18:15:21.402116Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-02T06:27:06.683926Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech.arXiv preprint arXiv:2006.04558, 2020","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2607.12706","last_updated":"2026-07-15T05:16:13Z","snapshot_observed_at":"2026-08-18T17:08:39.944329Z","submitted_at":"2026-07-14T12:28:43Z","title":"AutoSIFT: Automatic Style Sifting for Controllable Speech Generation with Arbitrary Style Infilling","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T06:27:06.683926Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2607.12706"},"observation_digest":"sha256:15460504ee3c585e700d30887dfc6b0c567ffc4724b16281f683d8e505694b2b","observation_id":"8b525661-3042-45e4-ad7a-d4be247ee97a","resolution":{"observed_at":"2026-08-02T06:27:06.683926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-01T11:33:12.513735Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2607.19859","last_updated":"2026-07-22T07:48:37Z","snapshot_observed_at":"2026-08-20T07:41:47.263723Z","submitted_at":"2026-07-22T07:48:37Z","title":"StellarTTS: Sparse Temporal Embedding for Low-Latency and Robust Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:12.513735Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2607.19859"},"observation_digest":"sha256:0c67d94a4e21afb90dd2715a82f4e7f920e16c5c0fc4b490d2cc84e67f856eb8","observation_id":"a0d3e1bf-75f0-4f90-b74b-e46458c72f28","resolution":{"observed_at":"2026-08-01T11:33:12.513735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-01T08:57:40.230200Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2607.20951","last_updated":"2026-07-23T06:04:10Z","snapshot_observed_at":"2026-08-14T19:57:11.361105Z","submitted_at":"2026-07-23T06:04:10Z","title":"Designed Vocalizations Dataset: Sound-Designed Human and Animal Voices for Non-human Voice Conversion","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T08:57:40.230200Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2607.20951"},"observation_digest":"sha256:d3f46507e8067a947467feb55caca82721eeb9244933819db9710547d0a740ba","observation_id":"d0c683b4-48b1-4d44-8145-b7b42e8bd38a","resolution":{"observed_at":"2026-08-01T08:57:40.230200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-07-31T15:08:48.177277Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.24430","last_updated":"2026-07-27T13:42:37Z","snapshot_observed_at":"2026-08-20T10:37:33.217527Z","submitted_at":"2026-07-27T13:42:37Z","title":"Let Me Look at You: Advanced Facial Expression Modeling for Conversational Speech Synthesis","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-31T15:08:48.177277Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2607.24430"},"observation_digest":"sha256:4925fd287072661065e5e4e1e1967e9cfcc52bb708e30572f3b30acba0858269","observation_id":"b241bb6a-636c-4942-a9b6-e30e605fe28d","resolution":{"observed_at":"2026-07-31T15:08:48.177277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-06T00:40:33.129335Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2608.00998","last_updated":"2026-08-02T04:54:12Z","snapshot_observed_at":"2026-08-15T06:31:49.907405Z","submitted_at":"2026-08-02T04:54:12Z","title":"Beyond One-Size-Fits-All: Personalized and Culturally Adaptive Emotional TTS via Interactive Optimization of Individual Emotion Perception Spaces","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T00:40:33.129335Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2608.00998"},"observation_digest":"sha256:d27bc328e0d3c9e350db34a347bea3779ea43a01871cdcec989b3921f824a219","observation_id":"b272a6a6-74fa-42e1-a810-a34233ee9888","resolution":{"observed_at":"2026-08-06T00:40:33.129335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-10T04:20:42.145891Z","title":"arXiv preprint arXiv:2006.04558 , year =","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2608.07462","last_updated":"2026-08-07T17:57:45Z","snapshot_observed_at":"2026-08-19T16:31:52.529349Z","submitted_at":"2026-08-07T17:57:45Z","title":"SemBridge: Semantic Token Anchoring for Continuous-Latent Autoregressive Speech Generation","version":1},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-08-10T04:20:42.145891Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2608.07462"},"observation_digest":"sha256:981ff3ecf0461f26e0ef98f3dedcafc777658fe7ec4fbd004ba03a989131bd52","observation_id":"b3081ec9-7616-4677-9e7a-06de3c16e19b","resolution":{"observed_at":"2026-08-10T04:20:42.145891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2006.04558/citation-record","integrity":"/paper/2006.04558/integrity","json":"/paper/2006.04558/citation-record.json","paper":"/paper/2006.04558"},"outbound":[],"paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","latest_version":8,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 68 inbound Pith citation observations for arXiv:2006.04558."}