{"as_of":"2026-08-06T12:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5ca0a3871bb4b4870f9f0a031c24664583c8be7d72fd3871de860fd3dfd91683","coverage":[{"denominator":128,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T06:06:41.348728Z","state":"measured"},{"denominator":164,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":164,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":64,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":64,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T11:34:09.568289Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T18:15:21.415875Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-07-06T20:06:30.732685Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T06:19:09.507440Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2412.10117"},"observation_digest":"sha256:00cd3f9255430e70d85fe69dd45ba9b2ad4b5c3514e99c75e9766a23394b6b7d","observation_id":"6c87e634-1e3f-4c4c-abf6-a5802d28e46d","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2505.17589","last_updated":"2025-05-27T07:48:34Z","snapshot_observed_at":"2026-07-06T21:29:06.781083Z","submitted_at":"2025-05-23T07:55:21Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T05:27:25.425188Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2505.17589"},"observation_digest":"sha256:b1db482527852468a059dc732adc8e6fb8e8c898f3ec2da4f3ccd7dac02f5f57","observation_id":"6c6c408e-c345-4eb9-acb5-498095597e23","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2507.09318","last_updated":"2026-04-14T14:21:41Z","snapshot_observed_at":"2026-08-02T11:31:42.381498Z","submitted_at":"2025-07-12T15:18:47Z","title":"ZipVoice-Dialog: Non-Autoregressive Spoken Dialogue Generation with Flow Matching","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T04:29:41.285194Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2507.09318"},"observation_digest":"sha256:16576d85c993e3742b7b875be7f0d06d85cc51569ca57c6c812ac565043e993a","observation_id":"79a70a04-4116-494e-9f57-3e4dbf87a9df","resolution":{"observed_at":"2026-05-19T04:32:03.653380Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-06T11:34:09.568289Z","title":"F5- tts: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.22612","last_updated":"2025-08-29T06:09:29Z","snapshot_observed_at":"2026-08-06T11:34:07.488878Z","submitted_at":"2025-07-30T12:31:11Z","title":"Adaptive Duration Model for Text Speech Alignment","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T11:34:09.568289Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2507.22612"},"observation_digest":"sha256:3d528107400d188a13f86b57d539eaaa7dba0e1b62eb14306987c72c2fd72c35","observation_id":"56b91bcd-6dbd-4bf0-b070-a152a73ffcc4","resolution":{"observed_at":"2026-08-06T11:34:09.568289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-06T06:04:29.398373Z","title":"Panda-70m: Captioning 70m videos with multiple cross-modality teachers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00733","last_updated":"2025-08-07T04:31:04Z","snapshot_observed_at":"2026-08-06T06:04:28.316296Z","submitted_at":"2025-08-01T16:03:57Z","title":"AudioGen-Omni: A Unified Multimodal Diffusion Transformer for Video-Synchronized Audio, Speech, and Song Generation","version":4},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T06:04:29.398373Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2508.00733"},"observation_digest":"sha256:5b33a4e7849457a87ede81000aa3f1b64bbabf617d0a6e52daed4814b1385adf","observation_id":"a9f25310-eeaf-42b5-a2f9-e18080661d87","resolution":{"observed_at":"2026-08-06T06:04:29.398373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-05T23:41:51.090161Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.04996","last_updated":"2025-08-08T01:59:26Z","snapshot_observed_at":"2026-08-05T23:41:48.793102Z","submitted_at":"2025-08-07T03:08:49Z","title":"REF-VC: Robust, Expressive and Fast Zero-Shot Voice Conversion with Diffusion Transformers","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T23:41:51.090161Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2508.04996"},"observation_digest":"sha256:402f2fc3c4f3497648a8ce08ddb707937d931e35e05b68aef3e57801e4b91504","observation_id":"68fb7631-fa48-4740-92ed-bf2752351d82","resolution":{"observed_at":"2026-08-05T23:41:51.090161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-05T21:23:56.183003Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.08890","last_updated":"2025-08-12T12:25:53Z","snapshot_observed_at":"2026-08-05T21:23:46.223428Z","submitted_at":"2025-08-12T12:25:53Z","title":"Transient Noise Removal via Diffusion-based Speech Inpainting","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T21:23:56.183003Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2508.08890"},"observation_digest":"sha256:8bc1c01ff05234da18fe9998f4b84a5fa7c8ff862509032671004e1d0a5dd5b2","observation_id":"a296521a-1a36-4701-984e-94ed130ce88a","resolution":{"observed_at":"2026-08-05T21:23:56.183003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-05T21:07:03.845459Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09389","last_updated":"2025-08-12T23:12:18Z","snapshot_observed_at":"2026-08-05T21:06:51.682221Z","submitted_at":"2025-08-12T23:12:18Z","title":"ProMode: A Speech Prosody Model Conditioned on Acoustic and Textual Inputs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T21:07:03.845459Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2508.09389"},"observation_digest":"sha256:de4661ade354399dc63d25d0d2766b420789aa8efc728374caf94a1e61a08158","observation_id":"8a8a0fbf-90ba-4799-b43e-99a25a9b2dc5","resolution":{"observed_at":"2026-08-05T21:07:03.845459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-05T16:01:52.541441Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19098","last_updated":"2025-08-26T14:59:30Z","snapshot_observed_at":"2026-08-06T08:55:29.264884Z","submitted_at":"2025-08-26T14:59:30Z","title":"CLEAR: Continuous Latent Autoregressive Modeling for High-quality and Low-latency Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T16:01:52.541441Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2508.19098"},"observation_digest":"sha256:abb75e71621cdf053769bb293dea40685e44d2ca449aa3fbd5001e2607a0403b","observation_id":"745110e8-993f-4339-8a4b-51e03547cfbb","resolution":{"observed_at":"2026-08-05T16:01:52.541441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-05T14:22:47.082941Z","title":"F5-TTS: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.21407","last_updated":"2025-08-29T08:27:17Z","snapshot_observed_at":"2026-08-05T14:22:45.383808Z","submitted_at":"2025-08-29T08:27:17Z","title":"DRASP: A Dual-Resolution Attentive Statistics Pooling Framework for Automatic MOS Prediction","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T14:22:47.082941Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2508.21407"},"observation_digest":"sha256:97c0465749ace6dc1ec32272c2dafaf2039782bd4f8aff327fb563ed5d97930c","observation_id":"c30e0ee2-aebe-45ec-8876-4811e67e03ed","resolution":{"observed_at":"2026-08-05T14:22:47.082941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-05T12:00:45.309565Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02020","last_updated":"2025-09-04T03:36:39Z","snapshot_observed_at":"2026-08-05T12:00:44.845727Z","submitted_at":"2025-09-02T07:06:46Z","title":"FireRedTTS-2: Towards Long Conversational Speech Generation for Podcast and Chatbot","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T12:00:45.309565Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.02020"},"observation_digest":"sha256:fe6b12824d56e6629041ffab5ee2e8d8bfbdf6a143146ff762140c3aae8cd5dc","observation_id":"cc062fb2-9d13-477c-897c-c469a86beab9","resolution":{"observed_at":"2026-08-05T12:00:45.309565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2509.04072","last_updated":"2026-04-21T15:32:24Z","snapshot_observed_at":"2026-07-06T22:23:48.679487Z","submitted_at":"2025-09-04T10:05:06Z","title":"Computational Narrative Understanding for Expressive Text-to-Speech","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T19:30:53.250247Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.04072"},"observation_digest":"sha256:7fb8ac4bd25cca60c9d5eb1dc95bb2ae335f3854f473340640c36a1cb8af9e29","observation_id":"4a78d0d3-ffe7-4e6b-ace3-6a0c7ed7d7a9","resolution":{"observed_at":"2026-05-18T19:31:47.040481Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T18:51:21.004855Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-04T18:51:12.536913Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:21.004855Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:3dfb6300c4a7396a3a07349252a12feb4ad011719a9f3a4456ee0ae465e82507","observation_id":"7a4f428d-ff2b-48bb-8435-8c428573856f","resolution":{"observed_at":"2026-08-04T18:51:21.004855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T19:11:59.493517Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09748","last_updated":"2025-09-11T12:32:08Z","snapshot_observed_at":"2026-08-06T08:55:28.245549Z","submitted_at":"2025-09-11T12:32:08Z","title":"DiTReducio: A Training-Free Acceleration for DiT-Based TTS via Progressive Calibration","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T19:11:59.493517Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.09748"},"observation_digest":"sha256:1262986462abf37102ef9b8ac5e249656c99805c8d86690d6608aaafb8cd5edf","observation_id":"ed3f9ca8-e5be-4851-8410-9a44505abc26","resolution":{"observed_at":"2026-08-04T19:11:59.493517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T17:10:04.032952Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.11084","last_updated":"2025-09-14T04:25:13Z","snapshot_observed_at":"2026-08-04T17:10:02.142026Z","submitted_at":"2025-09-14T04:25:13Z","title":"Length-Aware Rotary Position Embedding for Text-Speech Alignment","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T17:10:04.032952Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.11084"},"observation_digest":"sha256:a26fa90256f83df0d329b896b8a6ddbe5bf8ef63f2b5b471d02dc2a2cc7a526b","observation_id":"8abe8bc2-12fa-4844-b473-5600f0f41221","resolution":{"observed_at":"2026-08-04T17:10:04.032952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2509.20086","last_updated":"2026-05-08T13:18:21Z","snapshot_observed_at":"2026-07-06T22:30:41.141182Z","submitted_at":"2025-09-24T13:05:09Z","title":"OLaPh: Optimal Language Phonemizer","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-18T14:24:51.262622Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2509.20086"},"observation_digest":"sha256:b84209a62a6aa1d5aac9f3e0c48f7f711574f3272c24bea90705b36714422b2e","observation_id":"4f640102-5e34-46e4-8537-08aaf241db06","resolution":{"observed_at":"2026-05-18T14:26:28.197858Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T11:29:30.853473Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.04593","last_updated":"2026-06-08T05:49:30Z","snapshot_observed_at":"2026-08-06T08:55:28.794308Z","submitted_at":"2025-10-06T08:47:38Z","title":"UniVoice: Unifying Autoregressive ASR and Flow-Matching based TTS with Large Language Models","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-04T11:29:30.853473Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2510.04593"},"observation_digest":"sha256:415b162fef5852adfde3b282d505f53768429880db9df6419deee6a976b33155","observation_id":"33ab5622-c8eb-4cf2-a821-eedc99c37674","resolution":{"observed_at":"2026-08-04T11:29:30.853473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2510.19414","last_updated":"2026-05-11T03:38:57Z","snapshot_observed_at":"2026-07-06T22:33:48.680749Z","submitted_at":"2025-10-22T09:34:31Z","title":"EchoFake: A Replay-Aware Dataset for Practical Speech Deepfake Detection","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-18T05:01:57.044869Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2510.19414"},"observation_digest":"sha256:0ef1e33c62c7fa3aa5f2f7aede54b9cd884c05681725cceaca9aec9aff4921c7","observation_id":"625e4c1e-9328-459d-a0f0-3e4a73d85bb5","resolution":{"observed_at":"2026-05-18T05:02:23.389284Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-04T12:22:34.670840Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T19:24:56.057631Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2601.15621"},"observation_digest":"sha256:3869c9cd1b64c1441cebb41b6cdcde7b8c5e20d450cce69fede04ade15f60500","observation_id":"6802849f-fdff-40e8-99d3-261260535888","resolution":{"observed_at":"2026-05-16T19:24:56.082402Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2602.05449","last_updated":"2026-04-20T12:18:51Z","snapshot_observed_at":"2026-07-06T22:44:37.993093Z","submitted_at":"2026-02-05T08:45:08Z","title":"DisCa: Accelerating Video Diffusion Transformers with Distillation-Compatible Learnable Feature Caching","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T07:28:08.873654Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2602.05449"},"observation_digest":"sha256:f79c4b261eaaaa43d70e727bb7f32b16b9f7cdc517e87af3937aaccd8f69e4db","observation_id":"4f30ab23-c136-4ca9-a2b3-42d2b2e93d6e","resolution":{"observed_at":"2026-05-16T07:30:44.408615Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-02T21:26:58.384960Z","title":"F5- tts: A fairytaler that fakes fluent and faithful speech with flow matching.arXiv preprint arXiv:2410.06885, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.20360","last_updated":"2026-06-28T22:31:13Z","snapshot_observed_at":"2026-08-03T22:05:49.929640Z","submitted_at":"2026-02-23T21:06:35Z","title":"Momentum Guidance: Plug-and-Play Guidance for Flow Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T21:26:58.384960Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2602.20360"},"observation_digest":"sha256:77d071cbd45507af4c733fa30048ae660c91a99479995df4447691c0dcd061c4","observation_id":"28100a7f-b896-488a-9ba6-cfc2b263102e","resolution":{"observed_at":"2026-08-02T21:26:58.384960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.00688","last_updated":"2026-04-21T12:14:57Z","snapshot_observed_at":"2026-07-06T22:51:26.754746Z","submitted_at":"2026-04-01T09:45:51Z","title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T23:00:18.720371Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.00688"},"observation_digest":"sha256:0aab0de7218f5817fd54c935e92ab4842a1bac841a2524c27d3a3aead8315094","observation_id":"37f5b00f-d071-43bc-a879-d03417914af0","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.02374","last_updated":"2026-04-27T15:09:03Z","snapshot_observed_at":"2026-07-06T22:51:44.663367Z","submitted_at":"2026-03-31T20:35:26Z","title":"Evaluating Generalization and Robustness in Russian Anti-Spoofing: The RuASD Initiative","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-08T02:27:05.776290Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.02374"},"observation_digest":"sha256:185675f58e4b3e79b608695dd5f90b550d36b7dd72169dcb477ac1b5f6ab4ea7","observation_id":"fa5d5a27-28a4-4ba2-85c8-85e6b23575d1","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.08363","last_updated":"2026-04-09T15:27:22Z","snapshot_observed_at":"2026-07-06T22:57:26.678728Z","submitted_at":"2026-04-09T15:27:22Z","title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T17:15:28.918204Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.08363"},"observation_digest":"sha256:5e92fc11ddb78b7b40ff2aed4d7c2ce380dc56d7a93b9d56d06c57585d2cbed3","observation_id":"bf142335-2d34-4897-b53f-351548ac5e5b","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.11103","last_updated":"2026-04-24T05:22:49Z","snapshot_observed_at":"2026-07-06T22:59:36.571641Z","submitted_at":"2026-04-13T07:20:20Z","title":"ActorMind: Emulating Human Actor Reasoning for Speech Role-Playing","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-10T16:03:15.572657Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.11103"},"observation_digest":"sha256:e7a26721b75d6564d0a2652d7f7325ef66dca51832dc2921c3a8971db55a3a9a","observation_id":"95ab61d6-9068-438f-ac13-cf519b90c0db","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.12292","last_updated":"2026-04-14T05:03:57Z","snapshot_observed_at":"2026-08-06T05:09:28.270784Z","submitted_at":"2026-04-14T05:03:57Z","title":"CoSyncDiT: Cognitive Synchronous Diffusion Transformer for Movie Dubbing","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T15:46:58.010112Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.12292"},"observation_digest":"sha256:210fc7c76cbd31d788566a15e078b69fadbb9cb19230475dba9b8d10e06d79f0","observation_id":"fddd77c5-97cf-4285-a811-33f2c59fbeef","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.13229","last_updated":"2026-04-14T18:56:13Z","snapshot_observed_at":"2026-08-02T00:07:11.058562Z","submitted_at":"2026-04-14T18:56:13Z","title":"ProSDD: Learning Prosodic Representations for Speech Deepfake Detection against Expressive and Emotional Attacks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T13:31:12.802897Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.13229"},"observation_digest":"sha256:f68eedcadb16b3a053761ab8ebd2f9b9f3898071149eb8bd7ce3bbfcd29c7788","observation_id":"31716ffd-6c08-48dd-8ef0-3fc45e234fff","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.19055","last_updated":"2026-04-23T13:29:24Z","snapshot_observed_at":"2026-08-02T20:24:54.657148Z","submitted_at":"2026-04-21T04:01:36Z","title":"ATRIE: Adaptive Tuning for Robust Inference and Emotion in Persona-Driven Speech Synthesis","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T02:11:12.460695Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.19055"},"observation_digest":"sha256:85d94906dc8cee58528676c501191965cb5108e1ed9d5ae470ddbe9f64d4c718","observation_id":"44a8a20f-d1f7-4db4-8ac0-a524fcd1c024","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.21164","last_updated":"2026-04-27T17:57:41Z","snapshot_observed_at":"2026-07-06T23:07:47.244208Z","submitted_at":"2026-04-23T00:13:16Z","title":"MAGIC-TTS: Fine-Grained Controllable Speech Synthesis with Explicit Local Duration and Pause Control","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T13:49:09.386368Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.21164"},"observation_digest":"sha256:e93370156f0794cf788416d862c8811748570f184a09b7fc19205c54b5eb9528","observation_id":"66f91db6-cd7d-4f8c-b658-9046eec7917c","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.21481","last_updated":"2026-06-23T08:05:26Z","snapshot_observed_at":"2026-08-02T06:04:17.766216Z","submitted_at":"2026-04-23T09:44:03Z","title":"Preferences of a Voice-First Nation: Large-Scale Pairwise Evaluation and Preference Analysis for TTS in Indian Languages","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T22:07:47.847453Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.21481"},"observation_digest":"sha256:3f18400b38d761df7761e7b5649f85e2e27b11ac70038cd0f1bc57aae27e8a3d","observation_id":"5efca61d-fc14-483f-a8a5-b570aff7a329","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.22209","last_updated":"2026-04-24T04:26:04Z","snapshot_observed_at":"2026-07-31T01:17:34.846334Z","submitted_at":"2026-04-24T04:26:04Z","title":"UniSonate: A Unified Model for Speech, Music, and Sound Effect Generation with Text Instructions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-08T09:18:04.870414Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.22209"},"observation_digest":"sha256:9f9fb258528fe97f0c85666b367b90b6becc45099465634d47d0242e27e5e9eb","observation_id":"7d49c06d-cd33-45e5-b390-cde1a62975b1","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2604.25441","last_updated":"2026-04-28T09:50:01Z","snapshot_observed_at":"2026-08-02T09:52:11.123243Z","submitted_at":"2026-04-28T09:50:01Z","title":"Praxy Voice: Voice-Prompt Recovery + BUPS for Commercial-Class Indic TTS from a Frozen Non-Indic Base at Zero Commercial-Training-Data Cost","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-07T14:31:27.333704Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2604.25441"},"observation_digest":"sha256:f4332f1b6bfe01d5d9dad6562b6010c5392eec6c790b50cc47f0197294d82fca","observation_id":"3f88102a-d3f8-4e2b-bea1-dc1833fd02e2","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.00861","last_updated":"2026-04-21T10:34:41Z","snapshot_observed_at":"2026-07-06T23:14:11.211164Z","submitted_at":"2026-04-21T10:34:41Z","title":"Voice Mapping of Text-to-Speech Systems: A Metric-Based Approach for Voice Quality Assessment","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T01:07:50.243903Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.00861"},"observation_digest":"sha256:6cd54648c1df6aafaeea441e0cfdd49c42338902a43b65bf2f2aded709c491b7","observation_id":"26b8b0d4-5e92-495c-87f2-f791f2106e7d","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.02496","last_updated":"2026-05-04T11:45:39Z","snapshot_observed_at":"2026-07-06T23:15:32.349713Z","submitted_at":"2026-05-04T11:45:39Z","title":"Tibetan-TTS:Low-Resource Tibetan Speech Synthesis with Large Model Adaptation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-08T02:59:39.588729Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.02496"},"observation_digest":"sha256:cbad1dade76e81daa070ca99a3f411a02c3f5d122335dec924201d764e3d43fa","observation_id":"1ea04f1a-b303-4ce7-81a4-4590c918ac7c","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.09386","last_updated":"2026-05-10T07:24:55Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T07:24:55Z","title":"Kinetic-Optimal Scheduling with Moment Correction for Metric-Induced Discrete Flow Matching in Zero-Shot Text-to-Speech","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-12T03:04:44.050889Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.09386"},"observation_digest":"sha256:b52520c91afd9986f02eb3cae02b40cb5213f1339426919a47ddbdc2cce9acea","observation_id":"e09260d3-9114-4559-bf35-bcfada59cc96","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.12310","last_updated":"2026-05-12T15:57:12Z","snapshot_observed_at":"2026-07-06T23:24:03.984179Z","submitted_at":"2026-05-12T15:57:12Z","title":"Poly-SVC: Polyphony-Aware Singing Voice Conversion with Harmonic Modeling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T04:09:48.737629Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.12310"},"observation_digest":"sha256:fdd5708631dc05d7e1edd55a95859b221e1373226695891bf4ca35df1290f10f","observation_id":"ffe31654-4602-4adc-97d5-69d55aef05a1","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.14555","last_updated":"2026-05-14T08:32:38Z","snapshot_observed_at":"2026-08-02T19:29:24.447293Z","submitted_at":"2026-05-14T08:32:38Z","title":"Break-the-Beat! Controllable MIDI-to-Drum Audio Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-15T01:28:16.995989Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.14555"},"observation_digest":"sha256:51ef0608245e3ac0e27fd8733ad9fd3eaeb543cc479511c1b3bfeb3308e45843","observation_id":"7db77c5a-aa51-45d0-ac6d-f6fc51618eb5","resolution":{"observed_at":"2026-05-16T06:06:41.789827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.15984","last_updated":"2026-05-15T14:17:19Z","snapshot_observed_at":"2026-07-06T23:27:11.118592Z","submitted_at":"2026-05-15T14:17:19Z","title":"Beyond Content: A Comprehensive Speech Toxicity Dataset and Detection Framework Incorporating Paralinguistic Cues","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-19T18:36:34.351533Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.15984"},"observation_digest":"sha256:77a37fdbd812dc51d4bac863f30729d8874ea430cd97d6e60894212ed8ad10c6","observation_id":"cc24fda1-0762-4c78-853a-83af54cecc58","resolution":{"observed_at":"2026-05-19T18:37:42.801214Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.23859","last_updated":"2026-05-22T17:18:50Z","snapshot_observed_at":"2026-07-30T18:50:10.745264Z","submitted_at":"2026-05-22T17:18:50Z","title":"Natural Yet Challenging to Detect: Robust In-the-Wild TTS through EMA and Dual-Scoring Prompt Selection -- Submission for WildSpoof 2026 TTS Track","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-25T02:19:26.775357Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.23859"},"observation_digest":"sha256:d60518274d74adab31ac53876e851b3cd2e656c5fee9f95837b00b9df60c927d","observation_id":"c1ba00f5-d7b0-4a96-86ff-54f2c60366c4","resolution":{"observed_at":"2026-05-25T02:20:14.259206Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.24618","last_updated":"2026-05-23T15:01:28Z","snapshot_observed_at":"2026-08-03T19:22:55.968438Z","submitted_at":"2026-05-23T15:01:28Z","title":"FC-TTS: Style and Timbre Control in Zero-Shot Text-to-Speech with Disentangled Speech Representations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T12:20:43.383513Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.24618"},"observation_digest":"sha256:f1bb9560e80926eb75136e2aea54ecc19e1277c09893fe2700fd69f44f912429","observation_id":"c5733c39-a634-47a8-8395-b5968f033f06","resolution":{"observed_at":"2026-06-30T12:24:39.709349Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2605.30993","last_updated":"2026-05-29T08:27:57Z","snapshot_observed_at":"2026-08-02T00:55:15.607531Z","submitted_at":"2026-05-29T08:27:57Z","title":"SwanVoice: Expressive Long-Form Zero-Shot Speech Synthesis for Both Monologue and Dialogue","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T21:05:54.061395Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2605.30993"},"observation_digest":"sha256:c8455f4e2e580d78d5d60f12ae6bccd922c6a5f733642b086ae3d1b758078905","observation_id":"04100367-437a-4eaf-a515-afbc032ae967","resolution":{"observed_at":"2026-07-01T20:26:13.156506Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.00684","last_updated":"2026-07-01T12:15:25Z","snapshot_observed_at":"2026-08-01T03:51:14.605700Z","submitted_at":"2026-05-30T11:45:30Z","title":"Local Diagnostics of Continuous Normalizing Flow for Out-of-Distribution Detection","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T18:17:44.307688Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.00684"},"observation_digest":"sha256:f5e51e2ad969e8c7e39c7de577fc6786348b2c0dc8d129ffe9532944325ca850","observation_id":"fca80790-ff8d-46e6-bd37-84b3b2c5517b","resolution":{"observed_at":"2026-07-01T20:36:12.148206Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.00684","last_updated":"2026-07-01T12:15:25Z","snapshot_observed_at":"2026-08-01T03:51:14.605700Z","submitted_at":"2026-05-30T11:45:30Z","title":"Local Diagnostics of Continuous Normalizing Flow for Out-of-Distribution Detection","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-02T22:43:42.208048Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.00684"},"observation_digest":"sha256:8f852e858d9eddcb66c9660249996c1656ff42750730176b927df27ccee83340","observation_id":"ca2cf036-900e-4586-a4fc-d868ea73ac25","resolution":{"observed_at":"2026-07-02T22:47:25.335375Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.05889","last_updated":"2026-06-04T08:58:57Z","snapshot_observed_at":"2026-07-06T23:45:47.161248Z","submitted_at":"2026-06-04T08:58:57Z","title":"GLASS: GRPO-Trained LoRA for Acoustic Style Steering in Zero-Shot Text-to-Speech","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-27T23:53:55.385445Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.05889"},"observation_digest":"sha256:cc71158364b8dc48d526f6c568fab8bf9e4fb76729e275e792a00353f02a3fa7","observation_id":"baa741de-c517-4131-a48f-adffa0ecc70d","resolution":{"observed_at":"2026-07-02T15:27:04.899928Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.09141","last_updated":"2026-06-09T03:52:24Z","snapshot_observed_at":"2026-08-03T07:58:02.783857Z","submitted_at":"2026-06-08T07:39:26Z","title":"FlashTTS: Fast Streaming TTS with MTP Acceleration and X-pred Mean Flow Distillation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T15:15:10.060770Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.09141"},"observation_digest":"sha256:5486205cbea6cb8031e4830914c9f6ef72a866b5f184c94e34c9267e71e5b8cd","observation_id":"594969c4-8b0c-4670-b72f-172c0f77fe7b","resolution":{"observed_at":"2026-07-03T03:37:35.158085Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.09234","last_updated":"2026-06-08T09:07:23Z","snapshot_observed_at":"2026-08-05T17:27:24.038541Z","submitted_at":"2026-06-08T09:07:23Z","title":"End-to-End Training for Discrete Token LLM based TTS System","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:06.893507Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.09234"},"observation_digest":"sha256:e59aab5265f31e9b85c193dd8e830a445a81c2887ee0fff66d85cd55efa3450f","observation_id":"0210a7e3-3fac-4255-bc50-7475f8cbaa51","resolution":{"observed_at":"2026-07-03T03:27:34.864723Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.19325","last_updated":"2026-06-17T17:51:50Z","snapshot_observed_at":"2026-08-03T14:31:26.558590Z","submitted_at":"2026-06-17T17:51:50Z","title":"Reference-Driven Multi-Speaker Audio Scene Generation from In-the-Wild Priors","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T19:04:25.542629Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.19325"},"observation_digest":"sha256:d514a297c2fc30c8545bc2fbec102d24f8f54a5078b37d8c7243979714812a7e","observation_id":"bee5dda8-50fc-4493-a0e4-8f8a1d89b6bc","resolution":{"observed_at":"2026-07-04T02:49:24.873272Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.23335","last_updated":"2026-06-22T13:42:31Z","snapshot_observed_at":"2026-08-02T14:38:10.010771Z","submitted_at":"2026-06-22T13:42:31Z","title":"The Watermark Shortcut: How Provenance Marking Sabotages Audio Deepfake Detection","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T06:45:08.471913Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.23335"},"observation_digest":"sha256:0bfbcd2312a1cadbadf65a591f563df0aa9c610768be9bf83b758b12952800e8","observation_id":"645a0c5e-e4ef-406a-aefd-704cbcf3ac60","resolution":{"observed_at":"2026-07-04T12:29:51.994305Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.27380","last_updated":"2026-05-11T23:32:45Z","snapshot_observed_at":"2026-08-02T06:20:13.224586Z","submitted_at":"2026-05-11T23:32:45Z","title":"A Survey of Automated Presentation Coaching: Systems, Methods, and Open Challenges","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-06-30T22:06:35.036905Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.27380"},"observation_digest":"sha256:ddae2a11135f0c5f1032e68b1f21be8129da4ee55ce062dfbdee747260506001","observation_id":"e71f793e-d49b-4a8e-84e3-8902340d7684","resolution":{"observed_at":"2026-07-01T14:15:47.511667Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.30580","last_updated":"2026-06-29T17:22:27Z","snapshot_observed_at":"2026-07-07T00:04:25.385438Z","submitted_at":"2026-06-29T17:22:27Z","title":"MeloDISinger: Melody-Aware & Duration-Preserving Singing Voice Editing with Audio Infilling","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-30T03:20:05.125635Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.30580"},"observation_digest":"sha256:29e4f095ebd2b627fd46d37f3ad3bc4717c2a96ecdfb4eb6723f1eff63f4947d","observation_id":"f4333c2d-f6ed-498b-b44d-d4806be4ead3","resolution":{"observed_at":"2026-06-30T03:24:12.479645Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2606.31247","last_updated":"2026-06-30T07:24:10Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T07:24:10Z","title":"FlexiSLM: A Dynamic and Controllable Frame Rate Spoken Language Model","version":1},"reference_index":206,"source":"arxiv_source","source_observed_at":"2026-07-01T03:50:26.873406Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2606.31247"},"observation_digest":"sha256:bd11f94c74052495c676deecdfd524d3bc7ff43c808d22d6cd46817ed5e5fa44","observation_id":"86df8d3d-97bb-4123-871d-8241dbcddc23","resolution":{"observed_at":"2026-07-01T11:45:47.128085Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2607.01238","last_updated":"2026-05-01T17:46:18Z","snapshot_observed_at":"2026-08-03T13:23:27.654913Z","submitted_at":"2026-05-01T17:46:18Z","title":"SPARCLE: SPeaker-aware Aligned Representations via Contrastive Language Embeddings","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-04T01:21:02.597103Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.01238"},"observation_digest":"sha256:014e576856f4d6664e80fce44b96afcd8ef15ff7f48637ceaebbbde588a7a0d4","observation_id":"d806db55-8911-4940-b939-33017a980de0","resolution":{"observed_at":"2026-07-04T01:29:22.073256Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-12T08:16:37.711326Z","title":"F5-TTS: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02633","last_updated":"2026-07-02T14:40:21Z","snapshot_observed_at":"2026-07-12T08:16:37.458692Z","submitted_at":"2026-07-02T14:40:21Z","title":"GRAFT: Grafted Reference Audio for Fine-grained Pronunciation in Zero-shot Text-to-Speech","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-12T08:16:37.711326Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.02633"},"observation_digest":"sha256:104a0112068f1d9de660b228472ccfaeeeb3e1de96f222a8293f7a2c7769ab1a","observation_id":"894c7ba6-3a9b-4448-adf4-85d4262a94f7","resolution":{"observed_at":"2026-07-12T08:16:37.711326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-12T02:38:17.163398Z","title":"arXiv preprint arXiv:2410.06885 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03418","last_updated":"2026-07-03T15:27:17Z","snapshot_observed_at":"2026-08-04T03:15:34.868441Z","submitted_at":"2026-07-03T15:27:17Z","title":"DETECT-3B-Omni is Agnostic of Content and Demographics","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-07-12T02:38:17.163398Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.03418"},"observation_digest":"sha256:940b3cad6720af3a5d900cde3b43d436f2e2696be5decb80122798ff8aa65dd3","observation_id":"5d34f0f4-396a-4ccd-aeda-d7cc5a1d7f0a","resolution":{"observed_at":"2026-07-12T02:38:17.163398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-07-11T07:46:45.098798Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":1},"reference_index":226,"source":"arxiv_source","source_observed_at":"2026-07-07T23:59:38.702609Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:07e585dba675903e2fc8d4c97a5677e66ece1c24eed4d33dbc6f60f1c3645fdb","observation_id":"5d387887-6bbb-4190-8ccb-1624c7f6a54b","resolution":{"observed_at":"2026-07-08T00:04:22.329310Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-11T07:46:49.059192Z","title":"CoRR abs/2410.06885 (2024) , author=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-07-11T07:46:45.098798Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":2},"reference_index":226,"source":"arxiv_source","source_observed_at":"2026-07-11T07:46:49.059192Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:9041bbc5097e83098552ccaea2fdfa8a7aa107770af6918508abc6b09968b211","observation_id":"5d07cda7-e665-4a31-bcdb-218d6e89ce22","resolution":{"observed_at":"2026-07-11T07:46:49.059192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2410.06885","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-08T18:15:21.415875Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","venue":"eess.AS","work_id":"2a9a61a0-5461-4302-8659-788b84ecca31","year":2024},"citing_paper":{"arxiv_id":"2607.06054","last_updated":"2026-07-07T09:31:19Z","snapshot_observed_at":"2026-08-03T11:14:59.725228Z","submitted_at":"2026-07-07T09:31:19Z","title":"BlueMagpie-TTS: A Token-Efficient Tokenizer, Language Model, and TTS for Taiwanese-Accent Code-Switching Speech","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-08T18:09:22.207379Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.06054"},"observation_digest":"sha256:785a22f4b49348cfdc03319d57a5d000179d1c50120fb8fec5e78f7e63dec95d","observation_id":"b412697a-b251-4b4a-b94f-6f3556a123e7","resolution":{"observed_at":"2026-07-08T18:15:21.418131Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-13T05:10:26.667731Z","title":"arXiv preprint arXiv:2410.06885 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09134","last_updated":"2026-07-10T06:44:04Z","snapshot_observed_at":"2026-07-15T23:18:04.278291Z","submitted_at":"2026-07-10T06:44:04Z","title":"ReGen: Hierarchical Multi-Prompt Representation Generation for Efficient Waveform Diffusion Models","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-07-13T05:10:26.667731Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.09134"},"observation_digest":"sha256:f29923e623fc2949301c3aeb6b40dc34d328f150eb24616fbd062d13ada973fd","observation_id":"c71bf561-fdf8-43ae-9893-d5f3ed4cb1e2","resolution":{"observed_at":"2026-07-13T05:10:26.667731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-01T18:46:13.279565Z","title":"F5-TTS: A fairytaler that fakes fluent and faithful speech with flow matching","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18662","last_updated":"2026-07-19T11:20:25Z","snapshot_observed_at":"2026-08-01T18:46:12.871149Z","submitted_at":"2026-07-19T11:20:25Z","title":"Staged Depth-Pruning Distillation of a Flow-Matching Text-to-Speech Teacher: A Compact Hindi Speech Synthesizer","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T18:46:13.279565Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.18662"},"observation_digest":"sha256:0a6824beea6e5e738eef64b627e68a911b37166350dc3d8a0f9db13d531001bf","observation_id":"5552ee31-eff9-4fb4-b7d4-e900b9788094","resolution":{"observed_at":"2026-08-01T18:46:13.279565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-01T11:33:12.225012Z","title":"F5- TTS: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.19859","last_updated":"2026-07-22T07:48:37Z","snapshot_observed_at":"2026-08-04T15:26:57.843639Z","submitted_at":"2026-07-22T07:48:37Z","title":"StellarTTS: Sparse Temporal Embedding for Low-Latency and Robust Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T11:33:12.225012Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.19859"},"observation_digest":"sha256:b62f142d15e4445aff3e78db325f47c7b0728981e4a84f943aa981ebb28ecac3","observation_id":"559dbbb9-104c-4d86-b497-679ab7fe43a6","resolution":{"observed_at":"2026-08-01T11:33:12.225012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-07-31T23:35:21.520078Z","title":"F5-TTS: A fairytaler that fakes fluent and faithful speech with flow matching","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23938","last_updated":"2026-07-27T02:30:15Z","snapshot_observed_at":"2026-08-06T09:15:17.907620Z","submitted_at":"2026-07-27T02:30:15Z","title":"Qwen-Audio-3.0-TTS: Freely Controllable and Highly Robust Speech Synthesis with Multi-Stage Training Paradigm","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-31T23:35:21.520078Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2607.23938"},"observation_digest":"sha256:1ca40c9daccb304adfbbf2e59de105ab201f28ecce1eb3fe42a51d808d204e18","observation_id":"a17ef54d-5c2e-4f32-b7b6-52593258209a","resolution":{"observed_at":"2026-07-31T23:35:21.520078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T02:47:35.599102Z","title":"F5-TTS: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00011","last_updated":"2026-06-19T17:35:42Z","snapshot_observed_at":"2026-08-06T12:10:30.772069Z","submitted_at":"2026-06-19T17:35:42Z","title":"DLLM-TTS: Block Discrete Diffusion Language Model for Text-to-Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T02:47:35.599102Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2608.00011"},"observation_digest":"sha256:05ea5954cbc40a1a524c57bada84d1f94e1432af64f7cabdbd6eee8ad5a4b8ef","observation_id":"599d5572-5398-415a-9896-66d7c5a80a71","resolution":{"observed_at":"2026-08-04T02:47:35.599102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T20:44:55.874935Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01796","last_updated":"2026-08-03T07:04:51Z","snapshot_observed_at":"2026-08-06T12:19:04.514435Z","submitted_at":"2026-08-03T07:04:51Z","title":"Multi-Backbone Self-Supervised Ensembles for Audio Deepfake Detection and a Cross-Track Analysis of Generation-Detection Asymmetry","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T20:44:55.874935Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2608.01796"},"observation_digest":"sha256:b65f9869e0acb73e55fb27921f920dd10f576189569cf77c0d7b76ad920c2e12","observation_id":"22b08551-8d3f-4c2b-b52d-d8db65fe5228","resolution":{"observed_at":"2026-08-04T20:44:55.874935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-04T16:29:18.581793Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching.arXiv preprint arXiv:2410.06885, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-06T12:25:31.902555Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T16:29:18.581793Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:61e6492b4fa52a6e0f859a01ac40bc30836407ab02504306ee91b86232a214cd","observation_id":"3ebdeb05-2734-4491-b463-0b8deb33a428","resolution":{"observed_at":"2026-08-04T16:29:18.581793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.06885/citation-record","integrity":"/paper/2410.06885/integrity","json":"/paper/2410.06885/citation-record.json","paper":"/paper/2410.06885"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"693090f8-48cd-433f-957e-c5c483264d26","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:03aeb6e41a0798736b99ba7b3f54f4daca21a791e55d3ce2dff1aaa991d68ced","observation_id":"184cfbaf-e39d-48c6-83f9-6f6eff009aea","resolution":{"observed_at":"2026-05-16T06:06:41.621101Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning , pages=","venue":null,"work_id":"e38b79cf-3429-4fda-9658-7d665a341fe8","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:13be939d93c666c2e4a7f51df8947e300bea7a91c6f57d01aceed1b4ce683500","observation_id":"382fd074-effa-4c31-9af4-e448e24c724a","resolution":{"observed_at":"2026-05-16T06:06:41.689945Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"53d35174-0a7b-4232-8532-8dc58228c19c","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:c24f59a745504e1ba7257208a85c24d3b039434ed8c90b888e794620b17f7331","observation_id":"e4fc6877-2055-4306-bbe2-c7caf6cbca81","resolution":{"observed_at":"2026-05-16T06:06:41.692626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c743056b-566a-4d23-8d2e-98a9673b9039","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:e2feb58a2a1c73c1a27e91c00968be63c5f12e1f636b028f65fd4653ffddf6b8","observation_id":"8abdeed1-e280-44ca-a245-f800b747dad0","resolution":{"observed_at":"2026-05-16T06:06:41.695104Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2023 , organization=","venue":null,"work_id":"b36e2699-594f-4c72-9e58-849f2423a9ab","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:4fb437cfa1221716a41953973472a6e1446d4b300a3e78561788b2edd34dcaf0","observation_id":"e917f49c-d9aa-40ee-8f05-c23cacc64b03","resolution":{"observed_at":"2026-05-16T06:06:41.697698Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE/ACM Transactions on Audio, Speech, and Language Processing , volume=","venue":null,"work_id":"1f9ad95e-cf49-4918-8540-65fb825a58bc","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b5bce90201fb80ceab344b9a041689e58ee1d3cb026bbd958477329616dc2831","observation_id":"056f0b72-4f3f-4b58-8495-87fd015192c8","resolution":{"observed_at":"2026-05-16T06:06:41.700556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6a236501-cf0e-455d-b189-5c7e3e58b54a","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:fdbe0d003cace25a4e029a035f6cebcafbcd8dfed5ef00cdb11b0ed59fa0076c","observation_id":"2cb7ccd3-b14c-44fa-ac05-21b2498c54ec","resolution":{"observed_at":"2026-05-16T06:06:41.703242Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2011 IEEE international conference on acoustics, speech and signal processing (ICASSP) , pages=","venue":null,"work_id":"cd2576db-9f47-42d2-ace0-87cd277ef15f","year":2011},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:6a5071568ae5afee780407ef52ba424025b653a60375ff1f0150c0c7fa476a4b","observation_id":"7dfcc645-cf50-47d9-ba3c-7cdca229eb9e","resolution":{"observed_at":"2026-05-16T06:06:41.706882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T16:15:06.603231Z","title":"International conference on machine learning , pages=","venue":null,"work_id":"49ca566e-890e-44a7-b59d-f7a0424c9d40","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:1995afdf082abdee637427b143bc36dd190c88ab98d53139180df9be3a0a1e32","observation_id":"17cf1d67-ab72-411a-88a0-2b24f112b430","resolution":{"observed_at":"2026-05-16T06:06:41.710190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Libri-light: A benchmark for","venue":null,"work_id":"54adba7a-d8c0-4e20-847d-1a7c2de8fb6e","year":2020},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:c13aae174c774a4f7b450392cfc1c76bfe9b7bab27ed114fe1189ffa2a19e6b7","observation_id":"39698f26-7f55-4daf-a145-f8b823fab942","resolution":{"observed_at":"2026-05-16T06:06:41.712460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"58c6ce0d-063a-4c8c-b96e-543cec29570a","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:cbb642a89917dbbd35918d190e9c5dbe3b2e9d32b5b384bb6bbccf06cf86fe05","observation_id":"135f4a0e-01c2-4d0f-b501-ced8e810d3a4","resolution":{"observed_at":"2026-05-16T06:06:41.714908Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"06d93374-7dc5-4ce1-aff0-c243185f37f1","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:2c17f60aa05c430721a7155b959f99be8854c75530e7e6c93362a2ac9a5b8b35","observation_id":"cfe835e8-e1b0-420e-a0b4-e39d268909fb","resolution":{"observed_at":"2026-05-16T06:06:41.717497Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e4cd471a-103c-44b5-8e37-e9195876c83e","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:82ff7845c0e66faadfa54bad1b7592f4996a8e8c05ef4b0738d480af605066e8","observation_id":"49df7fdf-d8d3-4516-89cc-b724c44ebe24","resolution":{"observed_at":"2026-05-16T06:06:41.719930Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d4616d21-a74f-4cd9-a68d-7320115f0113","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:a5374ee456bfc510799d249600f43cc405bc7c95001e2eb3373ed206f0a33bb9","observation_id":"7e59a5f8-bd05-402d-ab2c-8d32986d0b71","resolution":{"observed_at":"2026-05-16T06:06:41.722469Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7e21bd88-e542-40c0-9740-a6b06d18b370","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:1bb3556d07c12a51841ff0b5da6c42cbd8ac2941d09bdc0dd1058b0fbf453831","observation_id":"e074e560-a14d-4148-80f0-018200c97e26","resolution":{"observed_at":"2026-05-16T06:06:41.724877Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"caa7eebd-60c7-4f06-8fa5-b0f0b107b31b","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:a0248dba5ab7883317b2cddbbb6ae239fd28c2366e99a6988256627b27b74a79","observation_id":"f084290c-b028-468b-a07a-085dc9dee7af","resolution":{"observed_at":"2026-05-16T06:06:41.727302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Librispeech: an","venue":null,"work_id":"0f6b2a3b-cac7-4f08-b921-4bcaa847ee65","year":2015},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:f4264e863bb560e8d04273cd6ff898d879a731d3e68106de37ce835072c327fd","observation_id":"fa4cbe13-13c7-4969-9e93-c12ef1895dea","resolution":{"observed_at":"2026-05-16T06:06:41.729514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Didispeech: A large scale","venue":null,"work_id":"00ef3346-499c-49c1-8724-198411328bee","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:8ecce044f0c90e2a9f17cc44273083e1a0451292351d46bf540ca016d118afc7","observation_id":"b71c7846-79c1-4cf1-b29a-bf15c7334010","resolution":{"observed_at":"2026-05-16T06:06:41.731705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"35d43e40-8924-4a3f-93d3-0542d92c8850","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:ce398a95177b40ad149c2fe1dd19e29ba5d2e189b2fba276cb882079ee42eea4","observation_id":"45b01911-5853-436d-8bf3-93e61038a9ab","resolution":{"observed_at":"2026-05-16T06:06:41.733867Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Biometrika , volume=","venue":null,"work_id":"d3f0009a-cf00-4a1e-a028-85550f0c06b3","year":1980},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:d142576f8d0a03538d24c3526037f3ac7b16c4b71b2f1922dc1ea081907a2e5b","observation_id":"62dce785-d519-44eb-8282-43ff8a5411dc","resolution":{"observed_at":"2026-05-16T06:06:41.736130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T03:08:01.568147Z","title":"Neurocomputing , volume=","venue":null,"work_id":"0cfd4ae7-837c-4c4f-bf6d-afc10f5a8ed1","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:39bf64005974548da153b7050351fab020c8f9c92b2f4770ceaf81a8ab7bf324","observation_id":"ae940ce3-e756-42e4-8b3a-a1ec40ed3592","resolution":{"observed_at":"2026-05-16T06:06:41.738636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T00:04:22.873515Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"33853a50-5794-41c9-a379-a0de8ee74b38","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:9edff3eb489c4f069f297af3084ade191ec389850267d93f34c7b62b6ac7f768","observation_id":"37f607f3-44bf-462c-92b5-4a4dca306cca","resolution":{"observed_at":"2026-05-16T06:06:41.741422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2024 , organization=","venue":null,"work_id":"c00754db-feb5-4a1f-bca3-32cdc13e6888","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:764e0916771bd0b1cd0eca3234d229d4a7ef559199e0679aed5dc58dd76e443c","observation_id":"e15fe887-ef0e-40cb-9282-9abb977f74b5","resolution":{"observed_at":"2026-05-16T06:06:41.743792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":"b768f6d6-f7fe-437e-94f3-5be8a8dd7a7b","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:0c379a31db127c4e4141b7cefec32f288e9aa31686c54d051a44119a642784e4","observation_id":"05122b77-7a9b-4555-93ce-0af5007b8c7a","resolution":{"observed_at":"2026-05-16T06:06:41.746045Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a169b05b-e29a-4e35-aaa6-22425535f059","year":2015},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:9295b51b7dcd7fd5383ecc921f14c36f8bb3f0208d5b391aeb846e792bd8fdbb","observation_id":"770432b2-838d-4cdf-803b-9fc64a7b259d","resolution":{"observed_at":"2026-05-16T06:06:41.748329Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T02:58:02.476681Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision , pages=","venue":null,"work_id":"549228b9-61a0-4bf1-bd0e-f81b79f72916","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:9cc8e7197b0a2a73aca91034f92480ffd4af5b3b8cc6565b398b596b735b65f2","observation_id":"138b71b1-ec08-46f3-80cf-9f2b003aa636","resolution":{"observed_at":"2026-05-16T06:06:41.751225Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T03:08:02.484080Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"2f872461-c98f-40a7-919b-53d4d9b3e586","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:eed9125b782857662326468b208437b5039903074382330ecc723432a774b7ae","observation_id":"31ef130b-2706-4813-829b-00db111ff3aa","resolution":{"observed_at":"2026-05-16T06:06:41.753787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"507335e0-38ea-4e19-852b-8f443f9db596","year":2018},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:12057c4a946d20c85fd5807c77ba48747c49b92a767e013b1a6d0df32b25eb7f","observation_id":"26ef8c1e-1280-469b-8d4b-4c349aea037b","resolution":{"observed_at":"2026-05-16T06:06:41.755955Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Understanding diffusion objectives as the","venue":null,"work_id":"851eee59-fb38-463c-aefe-c0653206984f","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:3295d210112516c02ba94cd6d2f9f3acc91c867ea3460f40f185da02d63c3f2b","observation_id":"0d87d880-78a0-4a57-8e8e-9750f59de282","resolution":{"observed_at":"2026-05-16T06:06:41.758285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0eb30b8d-832c-4609-8ea0-0eef073c571c","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:8c8ec4d5bf42ec9868f80f7179ec5482614f11e3ea6f4ffc41d0f34e97222693","observation_id":"411ac7fa-a3e5-430c-8ea6-16f88d4466dc","resolution":{"observed_at":"2026-05-16T06:06:41.760460Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2023 , organization=","venue":null,"work_id":"13631084-1e5d-4d12-9ec2-414cb42ee723","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:3b404f11c8eabe3de178aac9bb99dfb3975239a2fc7443ad6aa67c1cf6167837","observation_id":"82fbe785-5f99-4d63-9918-b609b8e65531","resolution":{"observed_at":"2026-05-16T06:06:41.762685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"70803092-8e2e-41fb-b507-7312dc9bcd4b","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:0d1bb5b4de278cbef5e91e54812ca4b1f68a3e697aa0bb8b74b184f871461c49","observation_id":"2f06da8a-d704-4014-94e6-3a8e4d9f3db4","resolution":{"observed_at":"2026-05-16T06:06:41.764958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3f15b6e6-5b46-4e0f-9255-7a7fbf88ed1c","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:5d26e1082f32b45cab6528f050bf0b1fb1e31ddd92e5090967464578e68c6158","observation_id":"d194696d-76e5-4924-b7a8-606dc2911204","resolution":{"observed_at":"2026-05-16T06:06:41.767403Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7f077c45-b634-4bdb-80ba-d31ca805bded","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:4147f157465a1f7d28a3ec893c9f929289c57cb5d607d51d3b265f3808075367","observation_id":"0bcea5b8-4d79-4284-a15f-dff3fff7690e","resolution":{"observed_at":"2026-05-16T06:06:41.769592Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6f7647aa-fa20-4153-9c41-3fc98a99184a","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:52d5db56049543cb9b15700f3e781ea7ae4ce75673ddc3d11effdfebdb6be645","observation_id":"92ff2a18-647b-443b-815b-43bc75fab998","resolution":{"observed_at":"2026-05-16T06:06:41.771813Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5da967aa-0b74-40c8-9d41-bbf88c237a70","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:a4acd0aecd4a1a1d43914fa5c871020e440887e4cda8d0025548d184213fe47f","observation_id":"020b22f3-fab8-4e07-a8f2-38449abeea6d","resolution":{"observed_at":"2026-05-16T06:06:41.773970Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"85eb599c-482d-4c97-991b-d959c51d0417","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:67ada409cad9902a0f636304eba3adbd3a0fa647a7fed124e668440bee09fcf7","observation_id":"16c00806-af5b-4d14-b8ef-94440a3eac18","resolution":{"observed_at":"2026-05-16T06:06:41.776463Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"932d8219-a940-461d-ac4e-cab2d0cd34d6","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:94178a7fca3151b9b81936723193df2332d895ad85aeecfb82f6a5c4b199ab01","observation_id":"26fcc227-5de7-4684-9e4b-5c7e426ab52e","resolution":{"observed_at":"2026-05-16T06:06:41.779064Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T02:58:01.546295Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"48172a5a-0dfc-45cf-9fdc-988f99c16450","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:2bb36c88ce2f1d8d7da72f1e98c094b4dec8f5f7e8fb2e87d6fefbd6e3d83261","observation_id":"a8ca92a8-3162-4f1d-b98f-8b6e6dd9887b","resolution":{"observed_at":"2026-05-16T06:06:41.781689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"cde362f8-068a-469b-87c2-416f111851dd","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:eb7407248a821104d0d859603219c89d09ac393b0d0ca343f3e7d7b54ce9b64e","observation_id":"6761d9de-d2c6-434a-beca-def5d7147d96","resolution":{"observed_at":"2026-05-16T06:06:41.783760Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T00:04:22.770521Z","title":"IEEE/ACM Transactions on Audio, Speech, and Language Processing , volume=","venue":null,"work_id":"17bf8810-eaf6-46be-9aea-77965f2ca634","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:fa7fbf9bad4a5af18c4004497028416e8324c7ec949b263cb5d96f5b4156b726","observation_id":"f2809e8c-20f6-40f2-9dc5-5c41d97b3318","resolution":{"observed_at":"2026-05-16T06:06:41.786425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"df472b20-5939-46b6-853b-bb3c3a3ea29c","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:8b4e1112d44c7e8f90d22f8ea3f08b0f33d0eeaa563bfcf8e08ceff4085abf45","observation_id":"83a51553-10c6-42cf-8c5d-aa8583ff4597","resolution":{"observed_at":"2026-05-16T06:06:41.788918Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ed8b6bb2-f031-48c6-bd40-e5762f49bfef","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b4c17c8ea8d4d77e167db9873f97c436638c9b43e8343c5233c35a59782fffa1","observation_id":"a16b1765-db14-4c51-b69d-5dae4b513d03","resolution":{"observed_at":"2026-05-16T06:06:41.572451Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b7b66766-6a32-407b-99b5-85129f8360aa","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:9f9880e063c0ccf5631322d850342803854ec4facea211cbb2ab9044d929988d","observation_id":"ec2f71ab-6c37-47e2-b0a6-a1cc055b37aa","resolution":{"observed_at":"2026-05-16T06:06:41.574606Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T00:04:22.737491Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"45d942b3-bf9f-4a91-8d51-e2112a99641c","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:81de8a2370001691577a2afbae0b3bc8b3f82208047d4f019e4ff77a5e0e4aec","observation_id":"00f85ccd-5fa9-412d-9468-55b2ddff56ab","resolution":{"observed_at":"2026-05-16T06:06:41.576800Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3c17f680-1e6c-4b6c-a307-33d80b11b89b","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:e33655b6256a03cc71a11f3057b6e03bde156442457cc77c6f9098ad9590af29","observation_id":"8cddbb06-1171-4eae-a20b-53a99ecf5a04","resolution":{"observed_at":"2026-05-16T06:06:41.579005Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning , pages=","venue":null,"work_id":"a5b60fea-b451-4077-8595-a0e220135a3d","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:085f48e1e8011a980a314acd034534c3c18a26e4abd9c1448af7e01917cf4fdd","observation_id":"e5b8928c-0cf0-4d15-92ce-8aed9ad4c250","resolution":{"observed_at":"2026-05-16T06:06:41.581164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7bc94811-d1b8-4fc6-b2ae-3db07a94d2c4","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:17a24226ec91ce1b7053236a695558ffdbe1e7bafa65e867965a8eae665771af","observation_id":"aa1bd310-0371-41ee-b7c9-6b4a5968e9d9","resolution":{"observed_at":"2026-05-16T06:06:41.583533Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"19814794-6f45-4b13-8ac2-311b4d487f15","year":2018},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:4584deeae1a4bbeae904c8a4a89dcebe5f55dde1534ad948b4cfaa34f0d76281","observation_id":"8c546d79-2c3b-48b3-bbff-1a5dc384d6b6","resolution":{"observed_at":"2026-05-16T06:06:41.585504Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T01:13:13.577979Z","title":"International Conference on Machine Learning , pages=","venue":null,"work_id":"4a0e6427-47c9-4664-9fb6-ea7b851dc972","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:5bcbe4d98a6b741747a8a755b969d6c9b33007b42d46157bb4c429febf598537","observation_id":"d95f7b5e-f10c-4fc4-b773-84ef826e13d4","resolution":{"observed_at":"2026-05-16T06:06:41.587791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the AAAI conference on artificial intelligence , volume=","venue":null,"work_id":"b4e79197-b63c-46dc-b275-194815c2178e","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b0bf5f8a05b372572319c89c8c9fc6c890aa39ea28c7aefe423560174152a551","observation_id":"ff238eba-802b-4d79-90ec-d0a0f4004ea4","resolution":{"observed_at":"2026-05-16T06:06:41.590030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T01:13:13.791903Z","title":"IEEE Transactions on Pattern Analysis and Machine Intelligence , year=","venue":null,"work_id":"0134caf5-760a-4635-8a89-439baf668c4e","year":null},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:dcb5353db64446b462a515d9341598b9255f2516bbf2e0ef5469a4d927184c01","observation_id":"b4f434d5-f068-4806-98d4-9ecfd0cf3bd2","resolution":{"observed_at":"2026-05-16T06:06:41.592109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":"2406.02430","doi":"10.48550/arxiv.2406.02430","metadata_source":"pith","pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","venue":"eess.AS","work_id":"6e88ee95-1133-4302-a142-cdf8f9456a8d","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:88358971ef7f6d49a058a914547b8e031de41d9e34efdca2a18b446e184a41e2","observation_id":"6000b29f-4584-435c-9497-184926710ea2","resolution":{"observed_at":"2026-05-16T06:06:41.513885Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.06670","last_updated":"2020-03-05T20:37:08Z","snapshot_observed_at":"2026-08-05T11:17:11.092255Z","submitted_at":"2019-12-13T19:22:44Z","title":"Common Voice: A Massively-Multilingual Speech Corpus","version":2},"cited_work":{"arxiv_id":"1912.06670","doi":null,"metadata_source":"pith","pith_arxiv_id":"1912.06670","snapshot_observed_at":"2026-07-07T20:34:09.715401Z","title":"Sorokin, and et al","venue":"cs.CL","work_id":"2215ac34-dbe8-4874-8438-39b788ed0cae","year":2019},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/1912.06670","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:e83ef3cb4029a1437e2fcbeeba8498ba45f2b6a6ee2981929b76a5a2fa53a243","observation_id":"0193b1ef-cf8c-4438-bfde-1bb999c3fee4","resolution":{"observed_at":"2026-05-16T06:06:41.464496Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3a6094e5-2d7a-4caa-b9ab-552aa30dec09","year":1980},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:4e2d394424c29fe9fdc37e951cd746d609040199694cec1cee49507c96995265","observation_id":"a50f5152-1a8d-4454-8575-9085bb31d957","resolution":{"observed_at":"2026-05-16T06:06:41.594217Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15835","last_updated":"2025-05-21T16:55:34Z","snapshot_observed_at":"2026-08-04T14:33:39.740706Z","submitted_at":"2024-07-22T17:51:53Z","title":"dMel: Speech Tokenization made Simple","version":3},"cited_work":{"arxiv_id":"2407.15835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.15835","snapshot_observed_at":"2026-07-03T23:59:06.138044Z","title":"dmel: Speech tokenization made simple","venue":null,"work_id":"b196131d-713c-44ae-bab4-4d52a3a93b48","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2407.15835","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:6bde974ab2361392152526ebba198e48e478d25e108125c10e20c28d2bbd78dd","observation_id":"bd46321d-de84-432d-9afc-2b9df90a991f","resolution":{"observed_at":"2026-05-16T06:06:41.561300Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b0302c3a-7f4a-4c16-bb78-99f02a85a02b","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:96495d6bebd5f761765b87edf7498055c555a0cd9ca41912c2334e27a0c80f17","observation_id":"8392d6dd-47e2-4862-9973-e4996aab7dcb","resolution":{"observed_at":"2026-05-16T06:06:41.596322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0543e4eb-e775-4ff3-af67-2a2d9a15ebcd","year":2018},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:d212f2c44bdac418f59d965b2242219f007aa6945692db023b6555da45c37391","observation_id":"73616e53-bf2f-45a7-b1ba-18f39d2029b0","resolution":{"observed_at":"2026-05-16T06:06:41.598357Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:039ad3f930b82b2ab86fad1f96cf0dbc9839e7c2b5e091975342f67c03888de8","observation_id":"251b1e65-964d-4550-9ac4-6742880289fb","resolution":{"observed_at":"2026-05-16T06:06:41.535511Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"547a37bd-91b5-49d9-a147-2f82a380698f","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:71189a886d725ccc6a9417e8c60126202bd20d8dcc08277caf0856655e448527","observation_id":"990c6a6c-abb9-401e-b739-6751c5e2b797","resolution":{"observed_at":"2026-05-16T06:06:41.600249Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.13438","last_updated":"2022-10-24T17:52:02Z","snapshot_observed_at":"2026-08-03T16:47:47.192907Z","submitted_at":"2022-10-24T17:52:02Z","title":"High Fidelity Neural Audio Compression","version":1},"cited_work":{"arxiv_id":"2210.13438","doi":"10.48550/arxiv.2210.13438","metadata_source":"pith","pith_arxiv_id":"2210.13438","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"High Fidelity Neural Audio Compression","venue":"eess.AS","work_id":"bc645d2d-e9f2-4cb8-9a6d-bd557bc7a258","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2210.13438","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:ba16a35838f2acdea803e8686fcf8e482b0f31da66cc2e959d454c9dda400028","observation_id":"6540a55b-bd22-4fc7-a56c-1406a6e6ec8e","resolution":{"observed_at":"2026-05-16T06:06:41.424218Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-19T16:22:25.658509+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T16:22:25.658509+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"7ba216d6-dccc-4f6d-8177-3ad7e3c94681","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:5b3a47aef1ce8d0ae1879741a9d90a44c7f22ebc7fe6ffc847c0269c534b060e","observation_id":"d0dfb163-bede-4b44-acf5-25899bfda549","resolution":{"observed_at":"2026-05-16T06:06:41.602094Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14321","last_updated":"2025-03-14T00:10:58Z","snapshot_observed_at":"2026-07-06T17:20:33.356092Z","submitted_at":"2024-01-25T17:19:01Z","title":"VALL-T: Decoder-Only Generative Transducer for Robust and Decoding-Controllable Text-to-Speech","version":5},"cited_work":{"arxiv_id":"2401.14321","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.14321","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V ALL-T: decoder-only generative transducer for robust and decoding- controllable text-to-speech","venue":null,"work_id":"b655aed2-a337-4456-9a6c-5b69e0555397","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2401.14321","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:ea46274d522fbec2875093c04ecb76c58973fc5721e4e0a908b49e90efd77fd7","observation_id":"0e2b1200-720a-4f80-9466-db9ffa8ffb45","resolution":{"observed_at":"2026-05-16T06:06:41.479977Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":"2407.05407","doi":"10.48550/arxiv.2407.05407","metadata_source":"pith","pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","venue":"cs.SD","work_id":"e5ad925a-4045-49b5-b301-208bcbf3eca8","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b0c51e3388ea3e0eda1bc993b7246a749ce4d9fc0c85ac6eb6ad19a41df2c7bf","observation_id":"3593ad95-210e-44da-9a38-a0c5fac028af","resolution":{"observed_at":"2026-05-16T06:06:41.487270Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18009","last_updated":"2024-09-12T17:45:37Z","snapshot_observed_at":"2026-08-04T03:12:21.293095Z","submitted_at":"2024-06-26T01:38:37Z","title":"E2 TTS: Embarrassingly Easy Fully Non-Autoregressive Zero-Shot TTS","version":2},"cited_work":{"arxiv_id":"2406.18009","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.18009","snapshot_observed_at":"2026-07-10T10:37:01.584595Z","title":"E2 tts: Embarrassingly easy fully non- autoregressive zero-shot tts","venue":"eess.AS","work_id":"ebf25130-33e5-4225-9048-5b7bb647cc86","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.18009","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:8d9a7395127aa4cc067751c7aee05fbbd6cbf1ed45009ae8756b5ecdc51b7001","observation_id":"f9b72dbf-3421-4be2-88ea-9ebd7141c765","resolution":{"observed_at":"2026-05-16T06:06:41.502207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9979de5a-0d2d-466c-8c2b-982091a5cab6","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:5414293265e4ab47dce6de2e1258f155323361775a2ccc92a565b3074f752015","observation_id":"76e26e0e-c924-4807-a742-e30414002df5","resolution":{"observed_at":"2026-05-16T06:06:41.603949Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00587","last_updated":"2024-12-20T11:22:01Z","snapshot_observed_at":"2026-08-05T15:59:56.280627Z","submitted_at":"2024-09-01T02:43:33Z","title":"FLUX that Plays Music","version":2},"cited_work":{"arxiv_id":"2409.00587","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.00587","snapshot_observed_at":"2026-07-08T00:04:22.357205Z","title":"FLUX that plays music","venue":"cs.SD","work_id":"da9012a3-8def-481f-a85f-851e63cab8f3","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2409.00587","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:63fc34033d1b8d4258103ead1c51b24a61c839e8a00e920871383be1ce394764","observation_id":"55cffd1f-1a25-4931-86e6-90ce62e2b721","resolution":{"observed_at":"2026-05-16T06:06:41.532023Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e550cf03-a655-49af-bf53-11b2b97ac099","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:c799c67f6ed92778367d556ffad2d431a2486efd767161153a6b1fc2c12fe4c9","observation_id":"39e0d087-6075-4aad-beaa-316ec414676f","resolution":{"observed_at":"2026-05-16T06:06:41.605786Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11013","last_updated":"2023-05-18T14:45:09Z","snapshot_observed_at":"2026-08-06T08:18:02.266041Z","submitted_at":"2023-05-18T14:45:09Z","title":"FunASR: A Fundamental End-to-End Speech Recognition Toolkit","version":1},"cited_work":{"arxiv_id":"2305.11013","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.11013","snapshot_observed_at":"2026-07-10T11:27:03.142196Z","title":"FunASR: A fundamental end-to-end speech recognition toolkit","venue":"cs.SD","work_id":"a1b4d167-dd83-49f9-8309-1e4d5492acf5","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2305.11013","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:a3e2370b0081b4c93327c25de8421ee229186e92d8c2daf10c32b21698b9ee40","observation_id":"f19d96cf-cee9-4b9d-a175-f7cc473c622b","resolution":{"observed_at":"2026-05-16T06:06:41.549207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03283","last_updated":"2025-04-11T07:36:53Z","snapshot_observed_at":"2026-07-06T19:10:48.519252Z","submitted_at":"2024-09-05T06:48:02Z","title":"FireRedTTS: A Foundation Text-To-Speech Framework for Industry-Level Generative Speech Applications","version":2},"cited_work":{"arxiv_id":"2409.03283","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.03283","snapshot_observed_at":"2026-07-04T08:29:41.338585Z","title":"Fireredtts: A foundation text-to-speech framework for industry-level generative speech applications","venue":null,"work_id":"436e9370-d40e-41da-b2ca-f56199732ea1","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2409.03283","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:bc6abaaeae925c4e26f0face03c81ce0607c9d87e4a049b53e332c2c4d8a67ec","observation_id":"d20dfc8c-8fda-41d8-81f3-2483d87ab40e","resolution":{"observed_at":"2026-05-16T06:06:41.552278Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"566a2801-15b4-4987-8cb1-0365df989d93","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:013a2108e8ae9fc63568587bc5e1cc70bcbd4ce93356c9289c90e8c5ba626b3b","observation_id":"d842a766-2845-4756-8cb9-753c6aa29881","resolution":{"observed_at":"2026-05-16T06:06:41.607964Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c01525f9-d569-4924-affc-d9f929301250","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:22a9948cd6b85162f2a9ea85976bc669b59dc1d50402d7dae9ecae255e96bdd9","observation_id":"5dc1fe4d-63f6-412b-88ee-90cb657f3cde","resolution":{"observed_at":"2026-05-16T06:06:41.609832Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07855","last_updated":"2024-06-12T04:09:44Z","snapshot_observed_at":"2026-08-05T04:07:33.037081Z","submitted_at":"2024-06-12T04:09:44Z","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","version":1},"cited_work":{"arxiv_id":"2406.07855","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.07855","snapshot_observed_at":"2026-07-03T23:19:03.868044Z","title":"Vall-e r: Robust and efficient zero-shot text-to-speech synthesis via monotonic alignment","venue":null,"work_id":"d46897fa-6b0f-4803-b02b-59b87aee1d4b","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.07855","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:6daf73f3ce8d29a0bb1229623d2d1e218f32e6cb50bb0480e81ffd3011cb159a","observation_id":"50e7f70c-2504-4599-a672-a57d5f920719","resolution":{"observed_at":"2026-05-16T06:06:41.450210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05361","last_updated":"2024-09-07T15:08:24Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T13:24:54Z","title":"Emilia: An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","version":3},"cited_work":{"arxiv_id":"2407.05361","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.05361","snapshot_observed_at":"2026-07-03T03:37:35.221767Z","title":"arXiv preprint arXiv:2407.05361 , year=","venue":null,"work_id":"1cf20033-c768-4fc5-bcdd-f566ab0b9231","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2407.05361","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:32486091eee53a565c453e7795fce75d3b2281b312d6f0c7cbaa1b12ca95a90f","observation_id":"0847645e-1357-4457-be39-54962b7a6d8c","resolution":{"observed_at":"2026-05-16T06:06:41.461429Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6e97dcc8-54a0-4c9b-9722-44d185d709ab","year":2020},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b6ce52e0b65fcd7982763e3220c8e3c2f5e5b6108da71691209c182aa2713fbc","observation_id":"90f4b4a3-c070-4b6e-a7af-7c7cb611bbc9","resolution":{"observed_at":"2026-05-16T06:06:41.611821Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12598","last_updated":"2022-07-26T01:42:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-07-26T01:42:07Z","title":"Classifier-Free Diffusion Guidance","version":1},"cited_work":{"arxiv_id":"2207.12598","doi":"10.1109/cvpr52733.2024.02494","metadata_source":"pith","pith_arxiv_id":"2207.12598","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Classifier-Free Diffusion Guidance","venue":"cs.LG","work_id":"acf2c588-c088-4a6c-938e-150ad7c666d7","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2207.12598","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:46892b72c7cb283511191b3870366944cbbf4f2b24246762b00002769a9e6b47","observation_id":"566aa25a-8a79-4cf7-a0bc-13dd147b3e14","resolution":{"observed_at":"2026-05-16T06:06:41.475519Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"5e26f7cd-58c3-4863-a3b9-200391c5cd3d","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:84d5192ef133679c8100f2c270a1d7c942b825af5d69de9174e2d9175ef68624","observation_id":"f32a933f-30a4-4a78-922c-264139a17116","resolution":{"observed_at":"2026-05-16T06:06:41.613796Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ca1b8610-c6d9-469c-875e-2384b33356c2","year":2017},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:643fac7edab8d41ef133a1d4dad564079c59a437fa5e8f39b2d5ba9d47b88a77","observation_id":"ebbb30ca-95a3-4d9e-a81a-06451c8c585b","resolution":{"observed_at":"2026-05-16T06:06:41.616081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03100","last_updated":"2024-04-23T08:38:03Z","snapshot_observed_at":"2026-08-05T18:34:08.469382Z","submitted_at":"2024-03-05T16:35:25Z","title":"NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models","version":3},"cited_work":{"arxiv_id":"2403.03100","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.03100","snapshot_observed_at":"2026-07-04T11:59:50.976455Z","title":"Naturalspeech 3: Zero-shot speech synthesis with factorized codec and diffusion models","venue":null,"work_id":"4ac4c208-eb65-46ea-b155-1f2e74819809","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2403.03100","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:ff1aa3ceaf3d409aafa371f0c08ae5c2592d299b7616935c0b75510f9e3d05a9","observation_id":"8d16dd5c-9eaf-459a-abd1-6dc409c68dfe","resolution":{"observed_at":"2026-05-16T06:06:41.494637Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"38554272-d434-4cf5-8372-23a91be8df56","year":2020},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:14d7e3839ff73653e85e10f154be15ebb0e3b7aa2a69a51875c6cce19ab2c51f","observation_id":"f73d03f8-4924-44a8-8e3b-917d82665098","resolution":{"observed_at":"2026-05-16T06:06:41.618130Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8f62a625-7943-4fb3-bf46-ed4bd29e249d","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:cf0685fb0f9592ecece0f7ba2b81a0af2c51aff8eb8683eb5e31ae341f0fe5b2","observation_id":"62abbae3-ad0e-4124-8f4f-33644ca81839","resolution":{"observed_at":"2026-05-16T06:06:41.687869Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"0abdbc8e-8fac-4fbe-a223-9cab0549bb63","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:df1234ba4bb8aeedcd8cf06aa936d8312b9ed72407a0ac2f099414c480071cb3","observation_id":"5bf03bf3-f29d-4228-b0d0-20fa8b84e7f8","resolution":{"observed_at":"2026-05-16T06:06:41.636734Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d32840f8-b26f-4b36-8fc1-286b97ef5965","year":2020},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:b1185be66a8094cfb974636a62de665e6c46f17570df0e47aff8a7292c428500","observation_id":"81efa80b-8c0e-4e10-9ea8-69379bd77a1c","resolution":{"observed_at":"2026-05-16T06:06:41.639483Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f7975744-71da-4baf-a748-d5116565ca02","year":2021},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:c454cb31f9baa9ad68690a598986157e7ddc733f2178cae4fbe7537e3d9bd17b","observation_id":"029a410c-f93a-49be-80d4-6883e245f025","resolution":{"observed_at":"2026-05-16T06:06:41.642324Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"000394b4-c13f-4e78-855b-13ca38a613d8","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:a6d9c5e988c0cfef29d718e1120b77d3d93569cc7ff30c4f98fb4be32192a395","observation_id":"61509622-b74e-468f-990e-15b303224801","resolution":{"observed_at":"2026-05-16T06:06:41.644737Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"e3b06a63-2007-4b20-8a2f-eee09a426178","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:4dc0a96127bb427a1e9c35e066a8d6fe6193fb1cd2f03b11e785ad0b602450da","observation_id":"4d3017f8-b17b-4dea-b9e5-562add6d2a9a","resolution":{"observed_at":"2026-05-16T06:06:41.647121Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11427","last_updated":"2025-02-17T17:34:45Z","snapshot_observed_at":"2026-07-06T18:32:06.826381Z","submitted_at":"2024-06-17T11:25:57Z","title":"DiTTo-TTS: Diffusion Transformers for Scalable Text-to-Speech without Domain-Specific Factors","version":2},"cited_work":{"arxiv_id":"2406.11427","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11427","snapshot_observed_at":"2026-07-03T12:58:08.367612Z","title":"Ditto-tts: Efficient and scalable zero-shot text-to-speech with diffusion transformer","venue":null,"work_id":"aa4b6638-dbe1-4c0d-a6c6-da476d39abf9","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.11427","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:fd0054dc702fee7bcb655ecc5b74c5eefcd4a9af30c64091defae859a950da65","observation_id":"de99c861-4fa9-4af9-b871-0aa3d7d9c1e7","resolution":{"observed_at":"2026-05-16T06:06:41.413952Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04658","last_updated":"2023-02-16T18:48:56Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:56:10Z","title":"BigVGAN: A Universal Neural Vocoder with Large-Scale Training","version":2},"cited_work":{"arxiv_id":"2206.04658","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2206.04658","snapshot_observed_at":"2026-07-04T05:39:39.351499Z","title":"Bigvgan: A universal neural vocoder with large-scale training","venue":null,"work_id":"953c2238-9c78-4b7c-a435-58768d959ff6","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2206.04658","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:19955d1dda9e508c1e7610e063d18b5eee319ad159bdfcb23735a90e56145e82","observation_id":"2739dbb6-bbdd-4204-ada2-6c2aa237b80e","resolution":{"observed_at":"2026-05-16T06:06:41.419455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"bf8cdf68-8d76-4849-9437-01d46453961b","year":2019},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:e02fc7d4e2bac578cdf5f2b24b919e5fb3ac3cdbe0f61c6470e8cd2b4b8fb1b9","observation_id":"0ed13fdd-c2d6-4ba0-a0fe-e40d3e846436","resolution":{"observed_at":"2026-05-16T06:06:41.649322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11838","last_updated":"2024-11-01T14:45:36Z","snapshot_observed_at":"2026-07-06T18:32:23.999019Z","submitted_at":"2024-06-17T17:59:58Z","title":"Autoregressive Image Generation without Vector Quantization","version":3},"cited_work":{"arxiv_id":"2406.11838","doi":"10.48550/arxiv.2406.11838","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.11838","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Autoregres- sive image generation without vector quantization.arXiv preprint arXiv:2406.11838","venue":"arXiv (Cornell University)","work_id":"154fdf62-5447-4dd1-bd62-d1188d2c4915","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.11838","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:05dcf0330077add91c78d0e754614203a4cbe6b5ce9829d5efdcd8dd51327fc0","observation_id":"125ddf1a-becc-4768-b27d-ad19755f2a1c","resolution":{"observed_at":"2026-05-16T06:06:41.430014Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02747","last_updated":"2023-02-08T15:46:05Z","snapshot_observed_at":"2026-08-02T18:24:58.914589Z","submitted_at":"2022-10-06T08:32:20Z","title":"Flow Matching for Generative Modeling","version":2},"cited_work":{"arxiv_id":"2210.02747","doi":"10.1038/s41467-024-47656-z","metadata_source":"pith","pith_arxiv_id":"2210.02747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flow Matching for Generative Modeling","venue":"cs.LG","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2210.02747","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:f66b04efd8ec32f61600ec6c8e4fe028e44f666bb0456344ede5daf9bdee2b29","observation_id":"331e86a3-f3da-4ae3-9738-c8e8b7bd7894","resolution":{"observed_at":"2026-05-16T06:06:41.435065Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05551","last_updated":"2024-06-08T18:57:13Z","snapshot_observed_at":"2026-07-06T18:27:36.903877Z","submitted_at":"2024-06-08T18:57:13Z","title":"Autoregressive Diffusion Transformer for Text-to-Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2406.05551","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05551","snapshot_observed_at":"2026-07-03T03:27:35.627140Z","title":"Autoregressive diffusion transformer for text-to-speech synthesis","venue":null,"work_id":"5585ee67-b148-4f36-896c-88d83418d9f6","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.05551","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:2ff3d623aa2ae1d7fb98bdacabb42a552f067f9e3dfdecf7a4407d9dd619dec9","observation_id":"a37e1d39-7169-43b7-b78e-807172d6bac5","resolution":{"observed_at":"2026-05-16T06:06:41.440978Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09351","last_updated":"2024-09-14T07:44:35Z","snapshot_observed_at":"2026-07-06T19:15:21.535851Z","submitted_at":"2024-09-14T07:44:35Z","title":"E1 TTS: Simple and Fast Non-Autoregressive TTS","version":1},"cited_work":{"arxiv_id":"2409.09351","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.09351","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"355d5c61-ea60-4250-a705-4e1707a74e03","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2409.09351","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:d6c9888170f7f1e6faa30349e6e6d258ac0c9af515057797b6f2d2260dde355f","observation_id":"75cd1411-76c2-4917-a511-a2d538ac4d89","resolution":{"observed_at":"2026-05-16T06:06:41.445688Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"922a00c8-81e4-4f51-966d-399ee094dfc4","year":2022},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:3ef78caaa837c9d815dcdf8c34cf4978db3ebaa20a5f3a7231541ff02187d23d","observation_id":"0f0822c1-aaa1-42e3-9779-3c74a398ece2","resolution":{"observed_at":"2026-05-16T06:06:41.651636Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":"1711.05101","doi":"10.1137/1.9781611972825.47","metadata_source":"pith","pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Decoupled Weight Decay Regularization","venue":"cs.LG","work_id":"07ef7360-d385-4033-83f7-8384a6325204","year":2017},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:d7b8a80dd251a8c31487f0c0e75432f9aed5747b16644425c6b3d3ab951eed87","observation_id":"c01e52fb-c86c-41d3-9724-b39fa8213131","resolution":{"observed_at":"2026-05-16T06:06:41.454756Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05763","last_updated":"2024-06-19T04:52:56Z","snapshot_observed_at":"2026-07-06T18:27:45.631190Z","submitted_at":"2024-06-09T12:32:42Z","title":"WenetSpeech4TTS: A 12,800-hour Mandarin TTS Corpus for Large Speech Generation Model Benchmark","version":3},"cited_work":{"arxiv_id":"2406.05763","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.05763","snapshot_observed_at":"2026-07-04T12:19:49.635938Z","title":"Wenetspeech4tts: A 12,800-hour mandarin tts corpus for large speech generation model benchmark","venue":null,"work_id":"d35e72e8-205a-40f9-b2a7-d9c5514c1e70","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.05763","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:c7875025000c861797c2065171b8f40a9eeeb19703b3b7c5238dc801aa8071bd","observation_id":"138ad91b-1082-4925-afe0-dcdaae9df868","resolution":{"observed_at":"2026-05-16T06:06:41.458352Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"37827e56-dbd2-4136-8082-f88186a431ef","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":124,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:46afd6105488ae0ad9f44940a6fc638ab06623cc474f6a85485da4861b900f12","observation_id":"75a063f3-3f23-413c-ab19-d24643d01d78","resolution":{"observed_at":"2026-05-16T06:06:41.653865Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"86c1b5e7-50e8-49e9-9fe6-34b1ceeb2329","year":2023},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:f90e96d1ff59fbdcc5c7582f1f321face57271df4df506abdfec64923d534e62","observation_id":"5c4f6778-c367-4528-9c16-244c161e9833","resolution":{"observed_at":"2026-05-16T06:06:41.656334Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08551","last_updated":"2025-05-27T05:07:56Z","snapshot_observed_at":"2026-08-05T05:08:56.833881Z","submitted_at":"2024-07-11T14:36:53Z","title":"Autoregressive Speech Synthesis without Vector Quantization","version":2},"cited_work":{"arxiv_id":"2407.08551","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.08551","snapshot_observed_at":"2026-06-29T19:03:52.201267Z","title":"Autoregressive speech synthesis without vector quantization","venue":null,"work_id":"575d89b8-5e74-44f6-819c-38e6e0c52a9f","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":126,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2407.08551","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:eebc7466e66a42579085a3c479d0e72f6c05d7298ab3bf68638a6da04d4e931f","observation_id":"e06e1dfd-3074-4fde-a7d2-ea6c240bb9b7","resolution":{"observed_at":"2026-05-16T06:06:41.468405Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12717","last_updated":"2024-09-19T12:41:30Z","snapshot_observed_at":"2026-07-06T19:18:08.790424Z","submitted_at":"2024-09-19T12:41:30Z","title":"NDVQ: Robust Neural Audio Codec with Normal Distribution-Based Vector Quantization","version":1},"cited_work":{"arxiv_id":"2409.12717","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.12717","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"adc3c7ad-27be-4cae-b2ae-6d1d662a8302","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":127,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2409.12717","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:326b602914e5930b79e3107c9308fdf11ff081cbbf95a1506ece1c3c74d2a1fc","observation_id":"1686f71e-2a9a-4f76-b18f-d723e9cd13db","resolution":{"observed_at":"2026-05-16T06:06:41.471859Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":49,"verified_exact":25,"verified_fuzzy":26},"total_outbound_references":128},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 100 of 128 outbound references and 64 inbound Pith citation observations for arXiv:2410.06885."}