{"as_of":"2026-08-09T02:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e3af22c091198fee10dd69de958be70f4f51ee57460bc4fe02add8226fde2424","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":35,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:19:56.312162Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T18:40:03.192889Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":139,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:0a87c5c08b99807b4dcb632ed344065d6d584784f21a1f92b7bad9044434d9b6","observation_id":"f47485fe-6dab-43e4-b8fe-365529144a4b","resolution":{"observed_at":"2026-05-16T06:06:41.517814Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2410.13720","last_updated":"2025-02-26T16:05:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-17T16:22:46Z","title":"Movie Gen: A Cast of Media Foundation Models","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-11T14:16:18.521699Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2410.13720"},"observation_digest":"sha256:a4df50479d594001c094001e0945724258ed9d0fee61a6c0181cc1ff6a375d85","observation_id":"0c322811-7749-48e4-bc4b-4936f82b6af3","resolution":{"observed_at":"2026-05-11T14:16:25.914341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T14:19:56.312162Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.ArXiv, abs/2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19462","last_updated":"2025-05-31T22:36:04Z","snapshot_observed_at":"2026-08-07T16:55:50.852560Z","submitted_at":"2025-05-26T03:35:44Z","title":"VoiceStar: Robust Zero-Shot Autoregressive TTS with Duration Control and Extrapolation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:56.312162Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2505.19462"},"observation_digest":"sha256:5e9ee2f81358fd1f1bc911843295f73ceb90bda176bdb0d915441d4bd32e2f11","observation_id":"d5407115-8935-4156-98a6-2901d3393e33","resolution":{"observed_at":"2026-08-07T14:19:56.312162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T12:12:56.445581Z","title":"Naturalspeech 2: Latent diffusion models are natu- ral and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00350","last_updated":"2025-05-31T02:23:38Z","snapshot_observed_at":"2026-08-07T23:23:36.425454Z","submitted_at":"2025-05-31T02:23:38Z","title":"DiffDSR: Dysarthric Speech Reconstruction Using Latent Diffusion Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:56.445581Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2506.00350"},"observation_digest":"sha256:5d4030ea8853eeabe086b3fd5f9d47dd05c765acdc2154406340487ab9a6e468","observation_id":"74c40840-3b67-4339-a24d-74fc4cd50c58","resolution":{"observed_at":"2026-08-07T12:12:56.445581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T12:03:37.184518Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00868","last_updated":"2025-06-16T12:09:09Z","snapshot_observed_at":"2026-08-07T20:37:23.293010Z","submitted_at":"2025-06-01T07:17:16Z","title":"Multiverse Through Deepfakes: The MultiFakeVerse Dataset of Person-Centric Visual and Conceptual Manipulations","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.184518Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2506.00868"},"observation_digest":"sha256:f121c9ea087e28f7346cdcde5c0d0bc0142e13d16ccb9ebd4a328cefaef54259","observation_id":"84115302-5a8a-4079-8810-af0dbd16bb2d","resolution":{"observed_at":"2026-08-07T12:03:37.184518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2506.23552","last_updated":"2026-05-15T04:47:28Z","snapshot_observed_at":"2026-07-06T21:49:33.849898Z","submitted_at":"2025-06-30T06:51:40Z","title":"JAM-Flow: Joint Audio-Motion Synthesis with Flow Matching","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T00:46:39.196042Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2506.23552"},"observation_digest":"sha256:4a9fe4270ee6824607fa27b04b0bda8ef1ae9a75c22594e36235cf2347d3e98f","observation_id":"4e224cf6-1374-4e0e-851a-8f60fdf62d64","resolution":{"observed_at":"2026-05-22T00:50:51.041243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T21:00:07.579047Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01348","last_updated":"2025-07-08T09:21:24Z","snapshot_observed_at":"2026-08-08T00:30:57.740737Z","submitted_at":"2025-07-02T04:30:23Z","title":"SpeechAccentLLM: A Unified Framework for Foreign Accent Conversion and Text to Speech","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T21:00:07.579047Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.01348"},"observation_digest":"sha256:e1ff8002bc695e923747d51a099fd1e8885677c0ea20f6b67ccd3eab9930bce6","observation_id":"ec02f16b-1a6e-4d7c-89f3-9d90ad4f2018","resolution":{"observed_at":"2026-08-06T21:00:07.579047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T15:48:21.184511Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14988","last_updated":"2025-07-20T14:48:48Z","snapshot_observed_at":"2026-08-08T10:46:13.682261Z","submitted_at":"2025-07-20T14:48:48Z","title":"DMOSpeech 2: Reinforcement Learning for Duration Prediction in Metric-Optimized Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:21.184511Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.14988"},"observation_digest":"sha256:25cacd3e6f51935ea0a56664efdf9418ffa8bbc25aea3645f39f49e85eff1c02","observation_id":"a96b3cfc-3d52-47d9-b621-0ef2835b38f1","resolution":{"observed_at":"2026-08-06T15:48:21.184511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T14:04:00.825953Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19835","last_updated":"2025-07-26T07:12:02Z","snapshot_observed_at":"2026-08-07T16:55:18.620169Z","submitted_at":"2025-07-26T07:12:02Z","title":"SonicGauss: Position-Aware Physical Sound Synthesis for 3D Gaussian Representations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T14:04:00.825953Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.19835"},"observation_digest":"sha256:9a8f234a7d077b4840b37bd314910a57f3279cd470fe0e757157c82f357fab31","observation_id":"3b20f27b-795a-4b6b-b9ef-dba358d4ba82","resolution":{"observed_at":"2026-08-06T14:04:00.825953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T11:22:27.121148Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.22746","last_updated":"2025-08-01T03:37:42Z","snapshot_observed_at":"2026-08-06T13:36:59.117683Z","submitted_at":"2025-07-30T15:03:36Z","title":"Next Tokens Denoising for Speech Synthesis","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T11:22:27.121148Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2507.22746"},"observation_digest":"sha256:1426a2893eae456d446c196660610bcafc63076f37ab2e36c295f2f187dd2853","observation_id":"faba2588-9e08-487e-b264-81415b3455e9","resolution":{"observed_at":"2026-08-06T11:22:27.121148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T05:17:06.401559Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.02038","last_updated":"2025-08-14T01:47:28Z","snapshot_observed_at":"2026-08-08T00:25:19.301459Z","submitted_at":"2025-08-04T04:08:22Z","title":"Marco-Voice Technical Report","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T05:17:06.401559Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2508.02038"},"observation_digest":"sha256:6fd7052cdc93378ec0d79b5f8ec2bad4ac9cc4d152520621e76e5438dfe204de","observation_id":"3dd2c9a9-4ed4-4bde-be05-e8b4c8dbfb64","resolution":{"observed_at":"2026-08-06T05:17:06.401559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T05:05:28.156213Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.02391","last_updated":"2025-08-04T13:17:49Z","snapshot_observed_at":"2026-08-08T02:50:04.116531Z","submitted_at":"2025-08-04T13:17:49Z","title":"Inference-time Scaling for Diffusion-based Audio Super-resolution","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T05:05:28.156213Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2508.02391"},"observation_digest":"sha256:e69cdc82aaeb1322c578cc36157668c189ef776592297b9ef719f6c1afd9ca64","observation_id":"e8139e27-ff8e-4e4d-86e3-ba24bdd26ffb","resolution":{"observed_at":"2026-08-06T05:05:28.156213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-05T15:10:31.804516Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.20379","last_updated":"2025-08-28T03:00:30Z","snapshot_observed_at":"2026-08-08T13:26:19.436875Z","submitted_at":"2025-08-28T03:00:30Z","title":"Audio-Guided Visual Editing with Complex Multi-Modal Prompts","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T15:10:31.804516Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2508.20379"},"observation_digest":"sha256:e9ed09f979954ff9a0d6e743532a58e36968f564417e095a02a4b342479f9f4e","observation_id":"6562b558-0a3e-403b-9f40-55481a3dceab","resolution":{"observed_at":"2026-08-05T15:10:31.804516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-05T13:35:01.874077Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero- shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00561","last_updated":"2025-08-30T17:10:22Z","snapshot_observed_at":"2026-08-08T00:31:06.181577Z","submitted_at":"2025-08-30T17:10:22Z","title":"FreeTalk:A plug-and-play and black-box defense against speech synthesis attacks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T13:35:01.874077Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2509.00561"},"observation_digest":"sha256:45da6947aa885961638f5807a14f945db25b91d359ce3d281f8b0496ad2d22f6","observation_id":"74f964eb-a931-415f-8f63-f15b96a40155","resolution":{"observed_at":"2026-08-05T13:35:01.874077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2509.18060","last_updated":"2026-04-18T03:04:41Z","snapshot_observed_at":"2026-08-08T00:01:52.882877Z","submitted_at":"2025-09-22T17:38:52Z","title":"TMD-TTS: A Unified Tibetan Multi-Dialect Text-to-Speech Framework for \\\"U-Tsang, Amdo and Kham Speech Dataset Generation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-18T14:25:40.217180Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2509.18060"},"observation_digest":"sha256:44c1c2c3a3f13e72fa535f930493a0a71e989268cd2c1bf8957aadebbaf0b25c","observation_id":"2ecbb115-ba05-42da-bc95-60a3ee269242","resolution":{"observed_at":"2026-05-18T14:26:27.981615Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-04T11:29:35.214098Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.04593","last_updated":"2026-06-08T05:49:30Z","snapshot_observed_at":"2026-08-06T08:55:28.794308Z","submitted_at":"2025-10-06T08:47:38Z","title":"UniVoice: Unifying Autoregressive ASR and Flow-Matching based TTS with Large Language Models","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-04T11:29:35.214098Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2510.04593"},"observation_digest":"sha256:7d8dcc229c281b9d8a2c942fae707150c541c22b5e290ab0782929b9040f18cc","observation_id":"5aa886f0-7a3f-46af-a0b0-bdfb212737bb","resolution":{"observed_at":"2026-08-04T11:29:35.214098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-04T12:22:34.670840Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T19:24:56.057631Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2601.15621"},"observation_digest":"sha256:ad522c2a1d344fb74e0227db8a3b60b9a750c108a345aa8a53950b3d42c5a292","observation_id":"2953e945-6422-45e4-b6ab-0d92a97d0c64","resolution":{"observed_at":"2026-05-16T19:24:56.145264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-03T00:11:16.339880Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.12304","last_updated":"2026-07-23T11:24:08Z","snapshot_observed_at":"2026-08-03T00:11:08.466564Z","submitted_at":"2026-02-12T03:25:41Z","title":"OmniCustom: Sync Audio-Video Customization Via Joint Audio-Video Generation Model","version":5},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-03T00:11:16.339880Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2602.12304"},"observation_digest":"sha256:fb2ba1f967cd40ed9c0c9f8bcd70e4cb2004da59cbcf063a3e4c52d21fe9c3f5","observation_id":"aa40bc84-52a8-4ae8-914d-109341bc5d0d","resolution":{"observed_at":"2026-08-03T00:11:16.339880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2604.06327","last_updated":"2026-04-07T18:05:28Z","snapshot_observed_at":"2026-07-06T22:54:48.916799Z","submitted_at":"2026-04-07T18:05:28Z","title":"A Novel Automatic Framework for Speaker Drift Detection in Synthesized Speech","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:24:54.627215Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.06327"},"observation_digest":"sha256:0541f91f833a6d417be28c97c67910ae1fd22085ae8fea01257ea4011ea3817c","observation_id":"b4843fa1-2926-4e71-8322-db573e7e4dc3","resolution":{"observed_at":"2026-05-11T00:41:00.491000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2604.11283","last_updated":"2026-06-01T08:50:38Z","snapshot_observed_at":"2026-08-02T05:26:21.584719Z","submitted_at":"2026-04-13T10:42:31Z","title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-10T16:36:33.264166Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.11283"},"observation_digest":"sha256:dfd2102308bc1cf0ba20aabe0b53c72c8148f688994820df8a3b1990024adb52","observation_id":"5e7439ab-4d74-46fe-b29a-8072b6e7f30e","resolution":{"observed_at":"2026-05-11T08:30:57.199763Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-12T22:04:31.302192Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.11283","last_updated":"2026-06-01T08:50:38Z","snapshot_observed_at":"2026-08-02T05:26:21.584719Z","submitted_at":"2026-04-13T10:42:31Z","title":"Multimodal Large Language Model-Enabled Video Translation: A Role-Oriented Survey","version":2},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-07-12T22:04:31.302192Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.11283"},"observation_digest":"sha256:a54845d29fb6d316646197faf0b5d54c52fa98584b2177e449b920cabcf1e999","observation_id":"46c4a353-49c8-4094-9e8c-fc49dce02c09","resolution":{"observed_at":"2026-07-12T22:04:31.302192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2604.24416","last_updated":"2026-04-27T12:45:18Z","snapshot_observed_at":"2026-07-06T23:10:29.508602Z","submitted_at":"2026-04-27T12:45:18Z","title":"Scaling Properties of Continuous Diffusion Spoken Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-08T03:43:10.132447Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2604.24416"},"observation_digest":"sha256:da110086c7f1dfbd5718c3a1d523b783da2b42b420fa60f7039fe15016ecee6d","observation_id":"2c603281-0c77-48ca-a921-0538cb1e2ce2","resolution":{"observed_at":"2026-05-11T21:56:28.356045Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2605.16578","last_updated":"2026-05-26T19:32:15Z","snapshot_observed_at":"2026-08-07T18:10:01.317172Z","submitted_at":"2026-05-15T19:32:28Z","title":"Voice \"Cloning\" is Style Transfer","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T18:58:45.582395Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2605.16578"},"observation_digest":"sha256:55c04d7799d6c1823ad3af6ab420eaed859819982ca41f1e8918897441257b07","observation_id":"648a05cf-8f2f-41f5-a8bc-d95b2bd7ed38","resolution":{"observed_at":"2026-06-30T19:05:01.038014Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2605.16964","last_updated":"2026-05-16T12:37:06Z","snapshot_observed_at":"2026-08-02T05:10:05.598125Z","submitted_at":"2026-05-16T12:37:06Z","title":"SemaVoice: Semantic-Aware Continuous Autoregressive Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T18:58:15.288299Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2605.16964"},"observation_digest":"sha256:74c41bbfb532e7944d1b0eee3b71798e427aa3e9b0f47876fbcc45cd285a8db0","observation_id":"b5a2da5f-67c4-4826-b9c7-27bf2f535ccc","resolution":{"observed_at":"2026-05-19T19:02:43.563073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.05852","last_updated":"2026-06-04T08:27:17Z","snapshot_observed_at":"2026-08-04T06:57:18.394840Z","submitted_at":"2026-06-04T08:27:17Z","title":"UniVoice: A Unified Model for Speech and Singing Voice Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T23:56:14.198308Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.05852"},"observation_digest":"sha256:3a6655f62cefd8cc996bad89537318e200d8a2bd7e3da1e26917ca102e451a99","observation_id":"f93442d7-5bf7-464a-b5f6-aadf112dbf63","resolution":{"observed_at":"2026-07-02T15:17:08.298843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.20650","last_updated":"2026-06-08T06:54:55Z","snapshot_observed_at":"2026-08-02T23:08:38.292740Z","submitted_at":"2026-06-08T06:54:55Z","title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T17:01:13.972071Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.20650"},"observation_digest":"sha256:c192cbdd15fcb5cf7a1f0af91eab3462b427995b600b4fffe6ff9af478807814","observation_id":"0b70f74c-f86a-40bd-aac6-a1530e6b36e5","resolution":{"observed_at":"2026-07-03T00:47:30.531681Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.23489","last_updated":"2026-06-22T15:35:30Z","snapshot_observed_at":"2026-07-06T23:58:11.694332Z","submitted_at":"2026-06-22T15:35:30Z","title":"MeshFlow: Mesh Generation with Equivariant Flow Matching","version":1},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-06-26T05:52:47.536542Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.23489"},"observation_digest":"sha256:44fe0b8f2e292c4b511f97316cfd6a236b902106c45c5b79978362e3abead0ca","observation_id":"99b1b84c-9075-47ee-a0fb-c22b443a5885","resolution":{"observed_at":"2026-07-04T12:49:52.683546Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.24320","last_updated":"2026-06-25T20:48:22Z","snapshot_observed_at":"2026-07-06T23:58:54.018557Z","submitted_at":"2026-06-23T08:57:34Z","title":"ZONOS2 Technical Report","version":1},"reference_index":190,"source":"arxiv_source","source_observed_at":"2026-06-25T22:37:15.072758Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.24320"},"observation_digest":"sha256:321686188a35491f75040f2e7ebfc20c843e094983ff731069591335ffa318e2","observation_id":"ef844789-06ca-4d27-aa25-2caf8243ba5c","resolution":{"observed_at":"2026-07-04T18:40:03.194270Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.24320","last_updated":"2026-06-25T20:48:22Z","snapshot_observed_at":"2026-07-06T23:58:54.018557Z","submitted_at":"2026-06-23T08:57:34Z","title":"ZONOS2 Technical Report","version":2},"reference_index":190,"source":"arxiv_source","source_observed_at":"2026-06-29T02:07:31.791835Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.24320"},"observation_digest":"sha256:3ab7cab620118fad4dab606d7572308da898b82d49354b8e45621ab61e67e03a","observation_id":"6f2f3709-6fd6-46c7-a37b-afc570b0240e","resolution":{"observed_at":"2026-07-01T18:15:59.065739Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":"2304.09116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-07-04T18:40:03.192889Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","venue":null,"work_id":"b19ab7ab-bff2-408d-a77d-b3e845d33277","year":2023},"citing_paper":{"arxiv_id":"2606.31247","last_updated":"2026-06-30T07:24:10Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T07:24:10Z","title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","version":1},"reference_index":150,"source":"arxiv_source","source_observed_at":"2026-07-01T03:50:26.873406Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2606.31247"},"observation_digest":"sha256:c5c739d780a8711d78a4364bb365c3c8f34636af32ab33937393ffd66c4e0ff1","observation_id":"78488924-78b0-442e-9b36-f069452bdea2","resolution":{"observed_at":"2026-07-01T11:45:47.268582Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-01T11:43:06.058605Z","title":"arXiv preprint arXiv:2304.09116 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.19810","last_updated":"2026-07-22T06:42:20Z","snapshot_observed_at":"2026-08-07T23:49:52.327639Z","submitted_at":"2026-07-22T06:42:20Z","title":"SimulS2ST-Omni: Data-Efficient Streaming Speech-to-Speech Translation via Explicit Trajectory Supervision","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-01T11:43:06.058605Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2607.19810"},"observation_digest":"sha256:96a51d94e34c7e9e1f474454bac37b5639d4985e99158b028f01ee48b07afbd4","observation_id":"f91965fd-1d8d-4831-aa05-a7edb4019bbb","resolution":{"observed_at":"2026-08-01T11:43:06.058605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-01T11:33:12.827054Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19859","last_updated":"2026-07-22T07:48:37Z","snapshot_observed_at":"2026-08-07T12:49:18.178302Z","submitted_at":"2026-07-22T07:48:37Z","title":"StellarTTS: Sparse Temporal Embedding for Low-Latency and Robust Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:12.827054Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2607.19859"},"observation_digest":"sha256:b6ff3b1722859a47569ffe53b2319500088c6b97c7bae67d83d0d8f6abc7596e","observation_id":"feb60fb7-d20c-49c9-93c2-cc2bafa5bac3","resolution":{"observed_at":"2026-08-01T11:33:12.827054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-06T00:40:31.445968Z","title":"Naturalspeech 2: Latent diffusion models are natu- ral and zero-shot speech and singing synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00998","last_updated":"2026-08-02T04:54:12Z","snapshot_observed_at":"2026-08-08T21:36:52.375893Z","submitted_at":"2026-08-02T04:54:12Z","title":"Beyond One-Size-Fits-All: Personalized and Culturally Adaptive Emotional TTS via Interactive Optimization of Individual Emotion Perception Spaces","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T00:40:31.445968Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2608.00998"},"observation_digest":"sha256:696a8eb3f55750eda193764a89fe9060d1684d4f7db1a9d94cacd2f651b664a5","observation_id":"43fc6afc-8c13-4382-8c7d-7ab8c9e49b12","resolution":{"observed_at":"2026-08-06T00:40:31.445968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-04T16:29:27.768440Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-07T23:09:45.300110Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-04T16:29:27.768440Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:736e823feb80a40587839abbb277b5d7d28efdb220aefee8083b89bad7871445","observation_id":"21be1899-e25c-4269-8f96-a72d1861f903","resolution":{"observed_at":"2026-08-04T16:29:27.768440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09116","snapshot_observed_at":"2026-08-07T00:14:47.793573Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers.arXiv preprint arXiv:2304.09116, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-07T23:09:45.300110Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T00:14:47.793573Z"},"links":{"cited_paper":"/paper/2304.09116","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:0333efb9676cd4829d71a21e4b0397004e6a5cafe0f9328d026087f2c30d5fd4","observation_id":"265648e2-a0c1-4985-8af3-e4ba1b4ce885","resolution":{"observed_at":"2026-08-07T00:14:47.793573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.09116/citation-record","integrity":"/paper/2304.09116/integrity","json":"/paper/2304.09116/citation-record.json","paper":"/paper/2304.09116"},"outbound":[],"paper":{"arxiv_id":"2304.09116","last_updated":"2023-05-30T16:09:10Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-08T00:30:28.663818Z","submitted_at":"2023-04-18T16:31:59Z","title":"NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 35 inbound Pith citation observations for arXiv:2304.09116."}