{"as_of":"2026-08-07T23:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b98b34cbe8e6de7608c1b66c2573b49af517d4008a9c874140d6f27f1f6c2ec7","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:45:17.467537Z","state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":16,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T07:39:23.947687Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T03:27:34.896349Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2507.09318","last_updated":"2026-04-14T14:21:41Z","snapshot_observed_at":"2026-08-02T11:31:42.381498Z","submitted_at":"2025-07-12T15:18:47Z","title":"ZipVoice-Dialog: Non-Autoregressive Spoken Dialogue Generation with Flow Matching","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T04:29:41.285194Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2507.09318"},"observation_digest":"sha256:f2cbc433a38d38810a0f5ebf14bb4b9fc9161f7437c704009cf008eafeabd6be","observation_id":"1f256cb3-22a5-4e2f-be92-73228fa1566d","resolution":{"observed_at":"2026-05-19T04:32:03.673585Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-15T12:21:48.698333Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.08977","last_updated":"2026-06-07T05:11:25Z","snapshot_observed_at":"2026-08-07T04:02:24.166463Z","submitted_at":"2026-03-09T22:11:40Z","title":"Universal Speech Content Factorization","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-15T12:21:48.698333Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2603.08977"},"observation_digest":"sha256:42f62b203eda377ad129cdd600aabf2f926e3e94a3e1b8c78cb4b2100dabb7ad","observation_id":"ab2e20d5-7d2d-4ad0-b454-c474d228d099","resolution":{"observed_at":"2026-07-15T12:21:48.698333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2604.00688","last_updated":"2026-04-21T12:14:57Z","snapshot_observed_at":"2026-07-06T22:51:26.754746Z","submitted_at":"2026-04-01T09:45:51Z","title":"OmniVoice: Towards Omnilingual Zero-Shot Text-to-Speech with Diffusion Language Models","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T23:00:18.720371Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2604.00688"},"observation_digest":"sha256:147109afa53b9f3396cd4d4f3df85158e8a87349933d209812f8b4f123e84036","observation_id":"ef4f0b28-14ef-48b4-aad7-45836011a635","resolution":{"observed_at":"2026-05-13T23:03:24.708672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2604.19679","last_updated":"2026-06-29T16:38:23Z","snapshot_observed_at":"2026-07-06T23:06:16.972438Z","submitted_at":"2026-04-21T16:57:23Z","title":"MMControl: Unified Multi-Modal Control for Joint Audio-Video Generation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T02:55:09.008954Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2604.19679"},"observation_digest":"sha256:6d8483f492ce5d2bd677db0bb3cd8aee2f41f2cee619ddf490a019e2ff233a9a","observation_id":"a7b0fc38-7d06-4288-9379-6e9f4c2ce3d3","resolution":{"observed_at":"2026-05-11T12:46:28.048637Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2605.08729","last_updated":"2026-06-29T11:45:55Z","snapshot_observed_at":"2026-08-01T08:29:29.120570Z","submitted_at":"2026-05-09T06:32:54Z","title":"Unison: Harmonizing Motion, Speech, and Sound for Human-Centric Audio-Video Generation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-12T02:28:14.734682Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2605.08729"},"observation_digest":"sha256:9c2cfb9f3a5afeab8b5f15b2b12b6f61a1c00f89f74412ea291f9c7c1051b47f","observation_id":"93f45acc-75fe-4625-954f-7155ac151c43","resolution":{"observed_at":"2026-05-12T07:36:45.127244Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2605.08729","last_updated":"2026-06-29T11:45:55Z","snapshot_observed_at":"2026-08-01T08:29:29.120570Z","submitted_at":"2026-05-09T06:32:54Z","title":"Unison: Harmonizing Motion, Speech, and Sound for Human-Centric Audio-Video Generation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-30T23:26:46.077894Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2605.08729"},"observation_digest":"sha256:3f0e582837629ea78e9b8138cece54069437e30c92df86bc094f841d31d0108c","observation_id":"a148795a-e901-425f-921a-39a80ed99b83","resolution":{"observed_at":"2026-06-30T23:35:07.821926Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2605.16026","last_updated":"2026-05-15T15:01:45Z","snapshot_observed_at":"2026-07-06T23:27:15.745896Z","submitted_at":"2026-05-15T15:01:45Z","title":"From Flat Language Labels to Typological Priors: Structured Language Conditioning for Multilingual Speech-to-Speech Translation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-20T19:25:18.488377Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2605.16026"},"observation_digest":"sha256:73330e6d96d9c5f9cf35ab22fa91ad1f945fe2c4fae370d0097d010036328777","observation_id":"fce86089-99f4-4aab-9876-524fa05de4cd","resolution":{"observed_at":"2026-05-20T19:28:55.049565Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2605.30993","last_updated":"2026-05-29T08:27:57Z","snapshot_observed_at":"2026-08-02T00:55:15.607531Z","submitted_at":"2026-05-29T08:27:57Z","title":"SwanVoice: Expressive Long-Form Zero-Shot Speech Synthesis for Both Monologue and Dialogue","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-28T21:05:54.061395Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2605.30993"},"observation_digest":"sha256:6513b41dd19cb7a715d61c1bd2b6e763e135a5fabc31f841e6331db98a6e0f19","observation_id":"456e7331-2449-4af2-a6c5-fa6a9ce4ec35","resolution":{"observed_at":"2026-07-01T20:26:13.093666Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2606.03455","last_updated":"2026-06-02T10:33:20Z","snapshot_observed_at":"2026-08-05T18:48:21.902772Z","submitted_at":"2026-06-02T10:33:20Z","title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","version":1},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-06-28T08:18:42.002083Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2606.03455"},"observation_digest":"sha256:75096e7957dac30e7c9d3541e1bf0ad72eaa743b2d551c6d9a15174d66e8caf6","observation_id":"35be436f-b14d-462d-9f24-c9059434f62d","resolution":{"observed_at":"2026-07-02T05:16:39.787929Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2606.06928","last_updated":"2026-06-05T05:43:15Z","snapshot_observed_at":"2026-07-06T23:46:37.903443Z","submitted_at":"2026-06-05T05:43:15Z","title":"VoxCPM2 Technical Report","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T21:18:22.911332Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2606.06928"},"observation_digest":"sha256:e3e15a134246fd077a15dfdc6b330582511e526e3d66e0ed8d73c265de344c60","observation_id":"d30edfa3-c12d-4673-9a20-d17c3e8d3216","resolution":{"observed_at":"2026-07-02T19:47:19.794924Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2606.07015","last_updated":"2026-06-05T07:59:17Z","snapshot_observed_at":"2026-08-07T10:25:26.085858Z","submitted_at":"2026-06-05T07:59:17Z","title":"Towards Unified Song Generation and Singing Voice Conversion with Accompaniment Co-Generation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T21:14:08.243894Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2606.07015"},"observation_digest":"sha256:1afaf7fd1b602bfdd9ae3880b75e6e5f6c03d42fa69998fd2778cfc0d705b558","observation_id":"d1c78e43-7078-41ac-acf0-683a37a18d96","resolution":{"observed_at":"2026-07-02T19:57:19.602737Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":"2506.13053","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-03T03:27:34.896349Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching","venue":null,"work_id":"f1ad0df1-58b6-4f8c-8c6f-ed9997868ca5","year":2025},"citing_paper":{"arxiv_id":"2606.09234","last_updated":"2026-06-08T09:07:23Z","snapshot_observed_at":"2026-08-05T17:27:24.038541Z","submitted_at":"2026-06-08T09:07:23Z","title":"End-to-End Training for Discrete Token LLM based TTS System","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T15:22:06.893507Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2606.09234"},"observation_digest":"sha256:cb59c5ab3a33915d552f0a0e2ed6d14c968ddf654e836ff9b4b089e82fed3d31","observation_id":"a83fb087-1b43-4394-b06b-ed907f6a3bd5","resolution":{"observed_at":"2026-07-03T03:27:34.898147Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-13T05:10:26.667731Z","title":"arXiv preprint arXiv:2506.13053 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09134","last_updated":"2026-07-10T06:44:04Z","snapshot_observed_at":"2026-08-07T15:40:45.553118Z","submitted_at":"2026-07-10T06:44:04Z","title":"ReGen: Hierarchical Multi-Prompt Representation Generation for Efficient Waveform Diffusion Models","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-07-13T05:10:26.667731Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2607.09134"},"observation_digest":"sha256:73c1e70291289afe80bc2ecf481e6d5401b84b76d07be642e4ea3a8c74e7a384","observation_id":"7a947619-156c-491b-a449-a4279363051c","resolution":{"observed_at":"2026-07-13T05:10:26.667731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-13T02:22:47.820537Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching.arXiv preprint arXiv:2506.13053,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09530","last_updated":"2026-07-22T14:29:04Z","snapshot_observed_at":"2026-08-06T08:33:56.882642Z","submitted_at":"2026-07-10T15:36:30Z","title":"FreyaTTS: A Compact Tokenizer-Free Flow-Matching Transformer for Turkish-First Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-13T02:22:47.820537Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2607.09530"},"observation_digest":"sha256:00d059b4b925f5d660d919d6af239ca45fb716a4249273f33e7878c46258d084","observation_id":"0297f999-eccb-423d-876a-5296308efd40","resolution":{"observed_at":"2026-07-13T02:22:47.820537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-08-02T07:39:23.947687Z","title":"Zipvoice: Fast and high-quality zero-shot text-to-speech with flow matching.arXiv preprint arXiv:2506.13053,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09530","last_updated":"2026-07-22T14:29:04Z","snapshot_observed_at":"2026-08-06T08:33:56.882642Z","submitted_at":"2026-07-10T15:36:30Z","title":"FreyaTTS: A Compact Tokenizer-Free Flow-Matching Transformer for Turkish-First Speech Synthesis","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-02T07:39:23.947687Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2607.09530"},"observation_digest":"sha256:f906e987543b05b57d28f567dd025df278dd246e0dab7c1cf6a7ec9a1b8d067a","observation_id":"e07bb246-f5cc-41a3-9d83-cc58e8eec40d","resolution":{"observed_at":"2026-08-02T07:39:23.947687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13053","snapshot_observed_at":"2026-07-30T22:26:14.049693Z","title":"ZipV oice: Fast and high-quality zero-shot text-to-speech with flow matching,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.26742","last_updated":"2026-07-29T10:33:56Z","snapshot_observed_at":"2026-08-07T04:02:27.388101Z","submitted_at":"2026-07-29T10:33:56Z","title":"Zero-Shot Face-to-Speech Synthesis via Latent Space Adaptation of a Style-Diffusion TTS Model","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-30T22:26:14.049693Z"},"links":{"cited_paper":"/paper/2506.13053","citing_paper":"/paper/2607.26742"},"observation_digest":"sha256:8661b367cc689ec1ad9cde344b35af66f75bc48b76b9cc6ead981d21628fefcf","observation_id":"51c0f031-9d9e-4949-9112-f3042e5cf35e","resolution":{"observed_at":"2026-07-30T22:26:14.049693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.13053/citation-record","integrity":"/paper/2506.13053/integrity","json":"/paper/2506.13053/citation-record.json","paper":"/paper/2506.13053"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:24.139390Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":"026974b6-e2dd-4bb3-b1af-13f6429056c1","year":2025},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.214738Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:1d690240aa3fa5e7220a9e8eeab6599e74131c556b4c0c5beda9287735b646e6","observation_id":"46dd6f12-388d-47f2-b4d5-1b58801c4773","resolution":{"observed_at":"2026-08-07T00:45:24.192871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:13.288815Z","title":"V oicebox: Text-guided multilingual universal speech generation at scale,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.288815Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:895379adc223edf1698f52f749983c78f17e363ee4f02059b19e4ec8d91d679c","observation_id":"2cbfe618-9751-4ea9-90ad-f4aac7513486","resolution":{"observed_at":"2026-08-07T00:45:13.288815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.967843Z","title":"E2 tts: Embarrassingly easy fully non- autoregressive zero-shot tts,","venue":null,"work_id":"494ded5b-752d-4f94-8c50-ee6132a65643","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.380779Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:9b08ed05c47c3f569539ec01e740af996fc8314b7bf91e998a9c5806292ab8fe","observation_id":"d4840c43-7c3b-4257-980b-e308b9d53508","resolution":{"observed_at":"2026-08-07T00:45:24.038565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-07T00:45:13.443312Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.443312Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:156d4ec63e05d2268ff43ccc736ed88f88399c392c09f27006a2592292f577ab","observation_id":"19e19724-5546-4746-8948-d7be6e8427d4","resolution":{"observed_at":"2026-08-07T00:45:13.443312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.827946Z","title":"MaskGCT: Zero-shot text-to-speech with masked generative codec transformer,","venue":null,"work_id":"b6a905fc-bc36-4036-b235-7a68db0f813e","year":2025},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.540878Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:eda7a8d50fa0895d4f026057a311800f8edd34deab8979fea1efea22c2c65286","observation_id":"c4ee4589-42be-40e8-b106-1220f2f921ad","resolution":{"observed_at":"2026-08-07T00:45:23.893891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.03283","last_updated":"2025-04-11T07:36:53Z","snapshot_observed_at":"2026-07-06T19:10:48.519252Z","submitted_at":"2024-09-05T06:48:02Z","title":"FireRedTTS: A Foundation Text-To-Speech Framework for Industry-Level Generative Speech Applications","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.03283","snapshot_observed_at":"2026-08-07T00:45:13.608161Z","title":"Fireredtts: A foundation text-to-speech framework for industry-level generative speech applications,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.608161Z"},"links":{"cited_paper":"/paper/2409.03283","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:35d2a359cd9d559d20a2a0202f17c4382637d98233242a4ea8631a4afc8133b5","observation_id":"47614f9a-f3b0-4357-b723-32f725028c9f","resolution":{"observed_at":"2026-08-07T00:45:13.608161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-07T00:45:13.675822Z","title":"Seed-tts: A family of high-quality versatile speech generation models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.675822Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:8b938ccd30c4510e2c498ba1113143d2fbea9fa3124031e237e6efcfb3a2fcda","observation_id":"811b01f7-0773-4610-a07c-7bf31ea74375","resolution":{"observed_at":"2026-08-07T00:45:13.675822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.721456Z","title":"Libritts: A corpus derived from librispeech for text-to-speech,","venue":null,"work_id":"82ec25f3-c536-448a-b7e0-b0844a4b4325","year":2019},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.755949Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:b50739835a89f5a5a567200e48064f0333dbbc1ec62ad514a0d34fdd832fe7c0","observation_id":"48f1086b-b0ef-4d23-aeb1-644adc8484d4","resolution":{"observed_at":"2026-08-07T00:45:23.776702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:13.847783Z","title":"Libriheavy: A 50,000 hours asr corpus with punctuation casing and context,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.847783Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:4c0fe002d9f193a33bbdfb78f31363826b7faa5213702d285a12a079029f8dfe","observation_id":"1a7c5959-0e17-4458-b945-020459b91c96","resolution":{"observed_at":"2026-08-07T00:45:13.847783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.572387Z","title":"Emilia: An extensive, multilingual, and diverse speech dataset for large-scale speech generation,","venue":null,"work_id":"b6dfdf78-abab-4955-9d4b-ba6367b02b99","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.919856Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:18753cb3fa864ccb92bc34b0d8d44079839dea818ef143babb733fa58c94041a","observation_id":"e190dbb3-1fbf-4a1f-9e04-e730369071e3","resolution":{"observed_at":"2026-08-07T00:45:23.630165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.407596Z","title":"Naturalspeech 2: Latent diffusion models are natural and zero- shot speech and singing synthesizers,","venue":null,"work_id":"d8d31510-68be-4218-9c0b-145dd6715cff","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:13.988814Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:30196110dce18bf8590a8dc23545140cea37c2fe0b4726fe1712bacbed1d1473","observation_id":"c4b312cb-f7a2-492f-b2c2-3ea22dbe1a6a","resolution":{"observed_at":"2026-08-07T00:45:23.479251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.269423Z","title":"Flow matching for generative modeling,","venue":null,"work_id":"9645650f-860a-4433-96c4-bacabc764fe2","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.064823Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:c26868428b357ff4ac6bbaa26de2f28dbfecf488adfdbe53bb03b299bd9b4a84","observation_id":"0648c2e9-ec80-40e7-b40c-854a16fed321","resolution":{"observed_at":"2026-08-07T00:45:23.333080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.141645Z","title":"Sf- speech: Straightened flow for zero-shot voice clone,","venue":null,"work_id":"31e36f61-8b2d-4f51-b407-b61a0e8f2c97","year":2025},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.143790Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:26a9e883c7bd9f8f264fe35a5f522051be118204777c7f1b1be8899a157d257c","observation_id":"c726225b-2ddf-465f-9cf5-1cc4fdbf401a","resolution":{"observed_at":"2026-08-07T00:45:23.200164Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:23.011553Z","title":"P-flow: A fast and data-efficient zero-shot tts through speech prompting,","venue":null,"work_id":"01a87716-e8e8-45a3-92d0-504ace453a45","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.242517Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:776498616d6ed1648f8a2e32f67e6cfe55a432835e4d151fe5043f84c9220e4b","observation_id":"f7b3f264-2b8f-494d-a920-a01aff7949ae","resolution":{"observed_at":"2026-08-07T00:45:23.080631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:14.325866Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.325866Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:b2bdcac1ac732e231aad677db8e47c49cfb49a1f7b7143eaae9b1ee18708bd65","observation_id":"8e3f0334-850f-41da-9b83-dbe729364cf0","resolution":{"observed_at":"2026-08-07T00:45:14.325866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:22.826311Z","title":"Zipformer: A faster and better encoder for automatic speech recognition,","venue":null,"work_id":"e29b7d23-cf0c-4b6b-a862-66d7bc67c8d2","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.417592Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:2f33e01f9d8586e52e33ad82661dae413f84011a64aaf15df245562201829d17","observation_id":"1e0b8f32-b682-4c29-ba91-3741fc249547","resolution":{"observed_at":"2026-08-07T00:45:22.897794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:22.699330Z","title":"Classifier-free diffusion guidance,","venue":null,"work_id":"958dc2fa-2dcd-4c8f-8510-345971685ffb","year":2021},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.504825Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:915e0208840f48f393e329d4ba06164d7d4cadab0976c8c33cba26376d3bc45d","observation_id":"2110306a-b124-47bc-be6c-8bd87d96343e","resolution":{"observed_at":"2026-08-07T00:45:22.760185Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:22.377824Z","title":"Freeu: Free lunch in diffusion u-net,","venue":null,"work_id":"a954f65d-503b-4031-bf16-317d901db185","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.572199Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:277a332f10c93e1567499e1fe3e599d5527484ed7d8912984b58a1caedb11897","observation_id":"170f12b1-dcbb-4192-bdf7-bbe89d575b1c","resolution":{"observed_at":"2026-08-07T00:45:22.605205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:22.144912Z","title":"U-dits: Downsample tokens in u-shaped diffusion transformers,","venue":null,"work_id":"3bdfab92-46ec-4c77-bf39-37164c4712d1","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.658013Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:abed4cfdd6d9b60151685e94875bf15ec39561b86b56c5ae4c16e9cc70b4e495","observation_id":"4a2bb12a-be56-4f7b-9cf2-5cc4415addbf","resolution":{"observed_at":"2026-08-07T00:45:22.210548Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:14.731869Z","title":"Fastspeech: Fast, robust and controllable text to speech,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.731869Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:88a2469f376cab9b6727ed1954ddc3241477979b4e266b8d2e72d6496e50fe0c","observation_id":"1fad3369-6cbe-4ed7-be2d-5ae2ffe0c242","resolution":{"observed_at":"2026-08-07T00:45:14.731869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:21.984329Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":"c67fd972-c5be-4715-976e-be980bd3346f","year":2020},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.832424Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:60b903cbc840fcdb1a441a0de18d80d832692c9b9a281cf375a9450be28d8909","observation_id":"94ec5498-4dba-42f3-ae84-7ae06ad8cc7c","resolution":{"observed_at":"2026-08-07T00:45:22.047617Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:14.930725Z","title":"Glow-tts: A generative flow for text-to-speech via monotonic alignment search,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:14.930725Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:47cb8a610d181b28a672b2a534e592d6442a62ca9dc0c60ee6800ba57aa197da","observation_id":"b9194bd8-2d98-4212-8bfe-fd2d1567bdf4","resolution":{"observed_at":"2026-08-07T00:45:14.930725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:21.794077Z","title":"Flow-tts: A non-autoregressive network for text to speech based on flow,","venue":null,"work_id":"6d52cc38-afaf-4761-9778-513035d1c65b","year":2020},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.033007Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:5aa373d9abe82bb96d52e4d5e0cf27cdb65c1bf26fe1b32b16beeabf3cd414e9","observation_id":"64bcbf28-fe5b-4eff-a317-75f4b2b627ca","resolution":{"observed_at":"2026-08-07T00:45:21.877248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:21.586788Z","title":"Simple- speech: Towards simple and efficient text-to-speech with scalar latent transformer diffusion models,","venue":null,"work_id":"7549ae34-c46e-4f00-969a-4f9dfdb07756","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.151321Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:a85614bbb25961c1b2bc91d37939f34dbd218b240b5226bbab12c933e4349406","observation_id":"a84aa383-803f-4188-93d9-7e3c6359d2ab","resolution":{"observed_at":"2026-08-07T00:45:21.707431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:21.378139Z","title":"DiTTo-TTS: Diffusion transformers for scalable text-to-speech without domain-specific factors,","venue":null,"work_id":"23a4b17e-d247-4be7-b43e-cee59158bcfd","year":2025},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.249852Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:567982bb4313a97a2ef63e483d4cfb2f1cb21e2be8911d87cd07154643997f65","observation_id":"e5c11d3b-3dc1-4568-b894-f926fe03ec62","resolution":{"observed_at":"2026-08-07T00:45:21.441890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:15.372934Z","title":"Convnext v2: Co-designing and scaling convnets with masked autoencoders,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.372934Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:0b539f7a11af9943871a0dda2ce470c1c06764f2524b372021c7e98ee82b21b2","observation_id":"9954c653-8243-48f0-8f7c-bee76df3b892","resolution":{"observed_at":"2026-08-07T00:45:15.372934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:21.132970Z","title":"On distillation of guided diffusion models,","venue":null,"work_id":"4214a1d4-08b8-4dc3-be6d-cdbf7bcd39e8","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.438075Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:9e80fa05545b7bf9c5a32f7f57870a6e99a79bbdd12b97c5dfdd46a78c2ae34d","observation_id":"38ae1325-a8b2-453f-a1ba-4bc4e81cc555","resolution":{"observed_at":"2026-08-07T00:45:21.229679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:20.939849Z","title":"Tacotron: Towards end-to- end speech synthesis,","venue":null,"work_id":"5985b625-d62b-49ca-ad06-e12ca158194e","year":2017},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.495205Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:3222578ea25af37d5214c24f5073d7cedb9e717721889e732c01cf315d99b573","observation_id":"8a9aeb93-b8ac-45a2-9e03-00166a481c25","resolution":{"observed_at":"2026-08-07T00:45:21.032849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:15.573777Z","title":"Natural tts synthesis by conditioning wavenet on mel spectrogram predictions,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.573777Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:7e0eefbc4ff35736e382758ef0b7bbef8777345613f151b3dfccf47f5136cf8c","observation_id":"c94e5b87-c82e-415e-b725-0232d8a39a93","resolution":{"observed_at":"2026-08-07T00:45:15.573777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.13066","last_updated":"2022-02-26T05:22:32Z","snapshot_observed_at":"2026-07-06T12:41:55.137989Z","submitted_at":"2022-02-26T05:22:32Z","title":"Revisiting Over-Smoothness in Text to Speech","version":1},"cited_work":{"arxiv_id":"2202.13066","doi":null,"metadata_source":"pith","pith_arxiv_id":"2202.13066","snapshot_observed_at":"2026-08-07T00:45:17.596121Z","title":"Revisiting Over-Smoothness in Text to Speech","venue":"eess.AS","work_id":"e35478f2-a3cf-47fa-8d7c-1075ca61b605","year":2022},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.639463Z"},"links":{"cited_paper":"/paper/2202.13066","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:277c1991b14ada99178c9a1907549c5bcebf3a2294aed380123a64e6a8bc632b","observation_id":"40a0df60-3838-450b-ae44-fdff04008690","resolution":{"observed_at":"2026-08-07T00:45:17.675731Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:20.681943Z","title":"Grad- tts: A diffusion probabilistic model for text-to-speech,","venue":null,"work_id":"6cf38217-c852-40b3-8cbb-30f3478599f9","year":2021},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.700325Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:6f38c1de35fe8a1ba26271b43d7d9157671a3311c54ffb31fd0aefc881dfdda7","observation_id":"e61555de-6d6c-4486-9be1-8c877b4ac97c","resolution":{"observed_at":"2026-08-07T00:45:20.775881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:20.500224Z","title":"Matcha-tts: A fast tts architecture with conditional flow matching,","venue":null,"work_id":"103f7f22-b862-4727-9f8f-56c04c47afdf","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.776492Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:5fe0feab065f7266d10354f605b0236ffd8006100a3dc724055b31df45a8a62f","observation_id":"52d9923c-d046-424b-b4af-68dee322933b","resolution":{"observed_at":"2026-08-07T00:45:20.589017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:20.287667Z","title":"Consistency models,","venue":null,"work_id":"d53e910f-b453-4497-a25d-c0aaddb3e18f","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.839694Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:68d5ff9a38d3b969167eb3cbd224737347140bb5dba046aede36a454a89db9a8","observation_id":"e5a4718e-8a2d-4a26-be94-da876beb5e8e","resolution":{"observed_at":"2026-08-07T00:45:20.391621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:20.043458Z","title":"Flow straight and fast: Learning to generate and transfer data with rectified flow,","venue":null,"work_id":"717b57bb-c942-43e2-af5c-b043c815b54e","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:15.926474Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:b6d83874196856286f7f90c3a7bfa5e89d6fb5a5e4b70d651a348a69762ce63c","observation_id":"b152212b-1d78-43b0-bb85-10d3a4ece214","resolution":{"observed_at":"2026-08-07T00:45:20.190716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:19.817615Z","title":"Comospeech: One-step speech and singing voice synthesis via consistency model,","venue":null,"work_id":"4d979b5e-3657-4132-9285-cfc04b2d3777","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.004306Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:ac0d9cf3a7ee839dd86895ee9521c13d75fd9dda9f31af12a9b2fe4c9c8edd97","observation_id":"f858a989-c3cc-448b-a3f4-7e1ede8df9f3","resolution":{"observed_at":"2026-08-07T00:45:19.931419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:19.660596Z","title":"Reflow- tts: A rectified flow model for high-fidelity text-to-speech,","venue":null,"work_id":"5d33fa76-cac7-4509-9f4d-9ceae7a990f6","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.052933Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:76ea82d3a372d7b30af7f449048aff6ef3b56da42707dc3e84b881cd548baae4","observation_id":"e52ae0af-4104-4462-99cb-d18419bc7542","resolution":{"observed_at":"2026-08-07T00:45:19.730383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:19.490708Z","title":"V oiceflow: Efficient text- to-speech with rectified flow matching,","venue":null,"work_id":"29e875a6-6dfe-4252-98e7-080cca2ebc1f","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.144964Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:e9624eb5a050aaaddc632a9aa3150310a60b4e693ec9be860a686724bc42bd26","observation_id":"da6a2ad3-611e-4da0-9196-71b71c128ca1","resolution":{"observed_at":"2026-08-07T00:45:19.567457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:19.248090Z","title":"Flashspeech: Efficient zero-shot speech synthesis,","venue":null,"work_id":"c5ec2385-c436-4cce-be98-0aafaee94149","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.189257Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:f84d416abe4f11e116f5af9ef70dc183fc013571d367bb1da823b1ed66ac0ed5","observation_id":"6e52ee7d-d6cf-49f8-a13f-9e2be46f25db","resolution":{"observed_at":"2026-08-07T00:45:19.354922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:19.017302Z","title":"Slimspeech: Lightweight and efficient text-to-speech with slim rectified flow,","venue":null,"work_id":"70d68874-e222-43da-8404-ea077692949b","year":2025},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.251832Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:099428d3d0754c63ff026cef988214eecdfa13a25f609194ba9defb29ffae6ad","observation_id":"fd3bdbbb-af80-4d3d-beef-ca010a89d28b","resolution":{"observed_at":"2026-08-07T00:45:19.136763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:18.824765Z","title":"Lightspeech: Lightweight and fast text to speech with neural architecture search,","venue":null,"work_id":"b67d6c99-d792-4cf0-8c11-00fd2e82fd6f","year":2021},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.341634Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:67361b3ccf238a98bb226d30f3bb7b212454865eacdcb0deda8cdf3d59cf7661","observation_id":"92523cae-b210-41cc-ae14-2d9e03326ba3","resolution":{"observed_at":"2026-08-07T00:45:18.922989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:18.653679Z","title":"Librispeech-pc: Benchmark for evaluation of punctuation and capitalization capabilities of end-to-end asr models,","venue":null,"work_id":"e61cb284-9c87-44e3-a6a7-0b4ac7997800","year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.410344Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:7a0e5a1426387f254d8cee91e10325bbb96cd18e3e522128b3792bbe090531eb","observation_id":"ceb84728-7a54-48fa-8c99-e7d41459232b","resolution":{"observed_at":"2026-08-07T00:45:18.724371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:18.491401Z","title":"Common voice: A massively-multilingual speech corpus,","venue":null,"work_id":"76a824ec-9d04-4c88-8114-fc6fb036ab8f","year":2020},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.499791Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:eeb275c8801a1fc91ef940a60637478936ee8b8b9c71828322fbb49983da5450","observation_id":"a08e7b59-6a11-42a9-b2db-21d3c47cefe3","resolution":{"observed_at":"2026-08-07T00:45:18.561055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:18.338576Z","title":"Didispeech: A large scale mandarin speech corpus,","venue":null,"work_id":"0e68afc9-4c2b-42ab-a304-3b2608222507","year":2021},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.572166Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:bbdbff9de7e23706f4aed70b9aba4e8f8d2c96e1a7829db3ae6adb0fcb180fb9","observation_id":"4bf2354c-ea56-44df-9c1f-b1f82f44ebda","resolution":{"observed_at":"2026-08-07T00:45:18.414015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:16.635875Z","title":"V ocos: Closing the gap between time-domain and fourier- based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.635875Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:b830163b3f9cff0f66359a60b6f45194bc38588ac4d7f736e7365c7cb7adc8c4","observation_id":"b922f287-09de-418f-95d3-a394b95e554c","resolution":{"observed_at":"2026-08-07T00:45:16.635875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:16.715337Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.715337Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:59f1fa9621187755515651560f9da2f19a06b43f34c6645fd1be253086ee6c54","observation_id":"66d38499-0d1c-4dba-9295-38bca924f20c","resolution":{"observed_at":"2026-08-07T00:45:16.715337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:16.775147Z","title":"Paraformer: Fast and accurate parallel transformer for non-autoregressive end-to-end speech recognition,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.775147Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:03603a70a720c4106b896f553f4a58e448010a6e41180672794565175bb956c3","observation_id":"28641d09-7533-4f30-9ff5-06211cc5fb0e","resolution":{"observed_at":"2026-08-07T00:45:16.775147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:16.855648Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.855648Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:345f419e648dbec58703e1ce72bf41ec528290f0d1772252384841a6b3494746","observation_id":"87dc4532-6b2a-4054-85d9-5caaf380e904","resolution":{"observed_at":"2026-08-07T00:45:16.855648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:16.920932Z","title":"Wavlm: Large-scale self-supervised pre- training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:16.920932Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:e479afbef00f89ede9176d121dba3d4d36c8aad0136bd9053d6494aba70142f8","observation_id":"bf4eb6b6-3f38-412d-a71e-47c23e715ee4","resolution":{"observed_at":"2026-08-07T00:45:16.920932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:17.031709Z","title":"Ecapa-tdnn: Em- phasized channel attention, propagation and aggregation in tdnn based speaker verification,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.031709Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:5ca4756ec1e9d8cd02c46dda377c0456d87b82cb64ac6526956bb6e027ee6806","observation_id":"523f1dfa-d805-481b-a534-f640a4ae254a","resolution":{"observed_at":"2026-08-07T00:45:17.031709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:18.135963Z","title":"Utmos: Utokyo-sarulab system for voicemos challenge 2022,","venue":null,"work_id":"d64e357e-616a-48cc-be42-eca6c83ee1f4","year":2022},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.111078Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:3ed6862cba62da64e6440dfa3c5ee87da0c37008e68b2e21ec717e046c7629a9","observation_id":"31e12026-c525-468c-a783-f51f3b05bb96","resolution":{"observed_at":"2026-08-07T00:45:18.197912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T00:45:17.180878Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to- speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.180878Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:0cc7d3262704d0fd5ed38af638dbf98fef26688b158aba9b8cf07f4639c817ef","observation_id":"263e86b0-e18e-426c-8237-e6dcd3010d8e","resolution":{"observed_at":"2026-08-07T00:45:17.180878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-07T06:02:40.568271Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-07T00:45:17.249552Z","title":"Cosyvoice 2: Scalable streaming speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.249552Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:951931ef4304c23507c17a2ec3e33a40100b8c79322ba8e0f9b9eb831523a26f","observation_id":"80c5d825-271b-4e00-9722-bc71a0a9a635","resolution":{"observed_at":"2026-08-07T00:45:17.249552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01710","last_updated":"2025-03-03T16:23:10Z","snapshot_observed_at":"2026-07-06T20:45:50.730195Z","submitted_at":"2025-03-03T16:23:10Z","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01710","snapshot_observed_at":"2026-08-07T00:45:17.314055Z","title":"Spark-tts: An efficient llm-based text-to- speech model with single-stream decoupled speech tokens,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.314055Z"},"links":{"cited_paper":"/paper/2503.01710","citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:c978de53d5836927274002f5e10df2bea62424ef5dc7616bbe6504a749b3081e","observation_id":"45102f47-a413-4735-b698-7e1c2858eaa9","resolution":{"observed_at":"2026-08-07T00:45:17.314055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:17.974932Z","title":"Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":"772a62f6-2574-4097-b260-57b56a71f09f","year":2022},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.387188Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:8e354080129d2f1fd3ebc0f2598f39bd8b30f4776c8b24fcad3b62adfeb05997","observation_id":"11b0aff7-79ae-4cc5-9a05-3e065b821938","resolution":{"observed_at":"2026-08-07T00:45:18.058741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:45:17.802325Z","title":"Amphion: an open-source audio, music, and speech generation toolkit,","venue":null,"work_id":"ff049626-5488-4018-b024-076e892a537c","year":2024},"citing_paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T00:45:17.467537Z"},"links":{"citing_paper":"/paper/2506.13053"},"observation_digest":"sha256:517bb0556a52c4b8a6fc6c795a86c9967b5136cc622da20f034a87e2fc1e4d7a","observation_id":"df6b7d1e-4c64-4752-ab52-b75bfd80e94d","resolution":{"observed_at":"2026-08-07T00:45:17.878084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.13053","last_updated":"2025-08-07T03:12:26Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-07T04:01:53.597585Z","submitted_at":"2025-06-16T02:48:17Z","title":"ZipVoice: Fast and High-Quality Zero-Shot Text-to-Speech with Flow Matching"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":1,"verified_fuzzy":35},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 16 inbound Pith citation observations for arXiv:2506.13053."}