{"as_of":"2026-08-08T06:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:00cccb99bad86ef923fdffbdb295c0e56100283315dd532dc748b06e4f236b28","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":52,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:58:40.546152Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T00:04:22.411558Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-16T06:06:41.348728Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2410.06885"},"observation_digest":"sha256:1a379d9d42750a8b962da9022cead9538052bb2afa99214102c9b06f53e4fbcb","observation_id":"251b1e65-964d-4550-9ac4-6742880289fb","resolution":{"observed_at":"2026-05-16T06:06:41.535511Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-07T06:02:40.568271Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T06:19:09.507440Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2412.10117"},"observation_digest":"sha256:a751f00ed02dc2052ac989126888da9083b06615794fa65dab8e894281c31cb9","observation_id":"5f83f5ca-41ac-4880-8515-d288a2c21a99","resolution":{"observed_at":"2026-05-13T06:19:09.545037Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T14:58:40.546152Z","title":"V ALL-E 2: Neural codec language models are hu- man parity zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16845","last_updated":"2025-05-22T16:10:01Z","snapshot_observed_at":"2026-08-07T14:51:56.686695Z","submitted_at":"2025-05-22T16:10:01Z","title":"Unlocking Temporal Flexibility: Neural Speech Codec with Variable Frame Rate","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:58:40.546152Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2505.16845"},"observation_digest":"sha256:d3648e8b740d3ec7f6ff46e886f7ac7ba17d046d6cb3241463c7cfc05c865171","observation_id":"91e21a39-50cf-492f-a160-8d31e37592d9","resolution":{"observed_at":"2026-08-07T14:58:40.546152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2505.17589","last_updated":"2025-05-27T07:48:34Z","snapshot_observed_at":"2026-07-06T21:29:06.781083Z","submitted_at":"2025-05-23T07:55:21Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T05:27:25.425188Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2505.17589"},"observation_digest":"sha256:63efeed47569d9a3b1bde08534069a7b922f881dc4b69fb436a5e037fcb119cd","observation_id":"59084775-ccb8-4363-aad0-0ed7a15b0c55","resolution":{"observed_at":"2026-05-16T05:27:25.492269Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T14:15:56.602314Z","title":"V ALL-E 2: Neural codec language models are hu- man parity zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19669","last_updated":"2025-06-02T10:03:25Z","snapshot_observed_at":"2026-08-07T14:06:47.459513Z","submitted_at":"2025-05-26T08:25:01Z","title":"Zero-Shot Streaming Text to Speech Synthesis with Transducer and Auto-Regressive Modeling","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:15:56.602314Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2505.19669"},"observation_digest":"sha256:d6dbc2bee80db03f0e46ffaf12cd162a3de9e8f9522705a5b3ee4fd579580ac8","observation_id":"f6446f42-6a4d-4da2-8133-f9a06b7a46a7","resolution":{"observed_at":"2026-08-07T14:15:56.602314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T13:23:05.343214Z","title":"Vall-e 2: Neural codec language models are hu- man parity zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.22054","last_updated":"2025-05-28T07:24:40Z","snapshot_observed_at":"2026-08-07T13:13:54.386349Z","submitted_at":"2025-05-28T07:24:40Z","title":"Voice Adaptation for Swiss German","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T13:23:05.343214Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2505.22054"},"observation_digest":"sha256:0cde56e042dc88fa2a5a4ba5632f750105729e118e252534c54895cc548c5346","observation_id":"78f0abd3-8987-4043-b332-5830629f06d2","resolution":{"observed_at":"2026-08-07T13:23:05.343214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T12:27:25.458653Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24496","last_updated":"2025-05-30T11:47:29Z","snapshot_observed_at":"2026-08-07T23:08:47.920163Z","submitted_at":"2025-05-30T11:47:29Z","title":"Speech Token Prediction via Compressed-to-fine Language Modeling for Speech Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:27:25.458653Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2505.24496"},"observation_digest":"sha256:c9eb11f70813c805b8dcbdac4713a5da3f091a29e20bc8814a84f49973946b8e","observation_id":"8bf7acdd-960c-4c94-9766-173297f31e72","resolution":{"observed_at":"2026-08-07T12:27:25.458653Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T12:12:56.370877Z","title":"Vall-e 2: Neural codec language models are hu- man parity zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00350","last_updated":"2025-05-31T02:23:38Z","snapshot_observed_at":"2026-08-07T23:23:36.425454Z","submitted_at":"2025-05-31T02:23:38Z","title":"DiffDSR: Dysarthric Speech Reconstruction Using Latent Diffusion Model","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:56.370877Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.00350"},"observation_digest":"sha256:cf847d3a0694d2ec35ead73f06ecc569dfd709d32b931abc24e9fa0727c6ca9c","observation_id":"170af19d-7e84-4a5d-bd61-fb515caa88a5","resolution":{"observed_at":"2026-08-07T12:12:56.370877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T11:58:33.743754Z","title":"VALL-E 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00975","last_updated":"2025-06-11T10:45:04Z","snapshot_observed_at":"2026-08-07T23:23:21.312419Z","submitted_at":"2025-06-01T12:01:40Z","title":"NTPP: Generative Speech Language Modeling for Dual-Channel Spoken Dialogue via Next-Token-Pair Prediction","version":4},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T11:58:33.743754Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.00975"},"observation_digest":"sha256:89b3bda3df2a0b9b4793212b18ed18384cad9a6fe6ebfe8325e7e89badf578e0","observation_id":"a3105a6d-23ac-44bc-adf1-d7dbaa5fd501","resolution":{"observed_at":"2026-08-07T11:58:33.743754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T13:17:52.088672Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05368","last_updated":"2025-05-28T09:13:41Z","snapshot_observed_at":"2026-08-07T20:52:22.030249Z","submitted_at":"2025-05-28T09:13:41Z","title":"Speaking images. A novel framework for the automated self-description of artworks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T13:17:52.088672Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.05368"},"observation_digest":"sha256:1bcc0d1b863f3a1634e4808f1d78835705ce481c06794af904a8720a775e77e6","observation_id":"65de2375-289d-467f-a1fe-5fcfac7c0ea8","resolution":{"observed_at":"2026-08-07T13:17:52.088672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T05:03:11.529192Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers.arXiv preprint arXiv:2406.05370, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08967","last_updated":"2025-06-13T10:07:42Z","snapshot_observed_at":"2026-08-07T20:32:05.075028Z","submitted_at":"2025-06-10T16:37:39Z","title":"Step-Audio-AQAA: a Fully End-to-End Expressive Large Audio Language Model","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:03:11.529192Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.08967"},"observation_digest":"sha256:d97d3a70bcb5067e1f7a5f51b3eb7d7e2d0f3b81b23167cc57fdf02479c00e9a","observation_id":"c04f6727-a3cd-412a-8844-60e880dc20d2","resolution":{"observed_at":"2026-08-07T05:03:11.529192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T04:43:22.311000Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.09847","last_updated":"2025-06-11T15:21:05Z","snapshot_observed_at":"2026-08-07T04:36:29.622203Z","submitted_at":"2025-06-11T15:21:05Z","title":"Dataset of News Articles with Provenance Metadata for Media Relevance Assessment","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T04:43:22.311000Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.09847"},"observation_digest":"sha256:55b3203afce740f00d1d4ef697b55b044df90d899c63d30024ea61a742780a20","observation_id":"bd381a56-5b71-410b-ad15-d4d22877f2c0","resolution":{"observed_at":"2026-08-07T04:43:22.311000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-07T00:15:44.715062Z","title":"VALL-E 2: Neural codec language models are human parity zero-shot text to speech synthesizers, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14767","last_updated":"2025-06-17T17:58:17Z","snapshot_observed_at":"2026-08-07T01:19:24.864953Z","submitted_at":"2025-06-17T17:58:17Z","title":"A Variational Framework for Improving Naturalness in Generative Spoken Language Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T00:15:44.715062Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.14767"},"observation_digest":"sha256:37ce4734a20dab8cbad5ad41f1ccf8a5b94dadd8983d60a36c6592035003de67","observation_id":"564ecf6b-fd82-4b41-b60e-b9bf15fb71bc","resolution":{"observed_at":"2026-08-07T00:15:44.715062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-06T22:19:03.126226Z","title":"V ALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22023","last_updated":"2025-06-27T08:45:21Z","snapshot_observed_at":"2026-08-07T19:17:21.433342Z","submitted_at":"2025-06-27T08:45:21Z","title":"Robust and Efficient Autoregressive Speech Synthesis with Dynamic Chunk-wise Prediction Policy","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T22:19:03.126226Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2506.22023"},"observation_digest":"sha256:21c1d65bbb5d76e57005e51793b0d20c50de6c7d6ac33dd3d68510689f0611c2","observation_id":"9a995078-6c79-4198-9ada-73e1eedc2b32","resolution":{"observed_at":"2026-08-06T22:19:03.126226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-06T15:48:20.513937Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14988","last_updated":"2025-07-20T14:48:48Z","snapshot_observed_at":"2026-08-08T06:04:11.949761Z","submitted_at":"2025-07-20T14:48:48Z","title":"DMOSpeech 2: Reinforcement Learning for Duration Prediction in Metric-Optimized Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T15:48:20.513937Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2507.14988"},"observation_digest":"sha256:686368457e866d9b3a14b28e647613003f7e459e59b87abea4dfadd71c8d29ec","observation_id":"c8be1b90-e219-4d50-826e-65bebd9db008","resolution":{"observed_at":"2026-08-06T15:48:20.513937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-06T11:22:25.548417Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.22746","last_updated":"2025-08-01T03:37:42Z","snapshot_observed_at":"2026-08-06T13:36:59.117683Z","submitted_at":"2025-07-30T15:03:36Z","title":"Next Tokens Denoising for Speech Synthesis","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T11:22:25.548417Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2507.22746"},"observation_digest":"sha256:eb2e88268516ebf83eb0908963fa483312d67e24f0a13a78b0d0ad7b095368cc","observation_id":"0c650765-46aa-4e7f-868f-ebc596c3d331","resolution":{"observed_at":"2026-08-06T11:22:25.548417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-05T16:01:52.534536Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19098","last_updated":"2025-08-26T14:59:30Z","snapshot_observed_at":"2026-08-06T21:25:08.912503Z","submitted_at":"2025-08-26T14:59:30Z","title":"CLEAR: Continuous Latent Autoregressive Modeling for High-quality and Low-latency Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T16:01:52.534536Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2508.19098"},"observation_digest":"sha256:d171137ef7cede35166f0b2745777a7d96b77df4bb3208d26c63447d71e6927a","observation_id":"53ce4747-da48-485a-b0f5-1a133c8439b7","resolution":{"observed_at":"2026-08-05T16:01:52.534536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-05T14:22:47.001462Z","title":"V ALL-E 2: Neural codec language models are hu- man parity zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21407","last_updated":"2025-08-29T08:27:17Z","snapshot_observed_at":"2026-08-07T08:26:07.109297Z","submitted_at":"2025-08-29T08:27:17Z","title":"DRASP: A Dual-Resolution Attentive Statistics Pooling Framework for Automatic MOS Prediction","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T14:22:47.001462Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2508.21407"},"observation_digest":"sha256:2dae5d99686533663882f6d91bf8592766f3c6990f9976bba747571809982884","observation_id":"986fe835-d4af-4a6d-886d-bc6c77efe5ff","resolution":{"observed_at":"2026-08-05T14:22:47.001462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T18:51:20.997584Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09631","last_updated":"2026-06-17T16:16:04Z","snapshot_observed_at":"2026-08-06T15:26:13.717443Z","submitted_at":"2025-09-11T17:16:52Z","title":"DiFlow-TTS: Compact and Low-Latency Zero-Shot Text-to-Speech with Discrete Flow Matching","version":5},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T18:51:20.997584Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2509.09631"},"observation_digest":"sha256:2aa8f5b2f66ba21017ae4ffb43089c0198de50f601d8c85ab9bf583df19a13b7","observation_id":"4dbb91b5-f808-4ded-a2db-5a1266f0324c","resolution":{"observed_at":"2026-08-04T18:51:20.997584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T19:11:59.490276Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09748","last_updated":"2025-09-11T12:32:08Z","snapshot_observed_at":"2026-08-06T08:55:28.245549Z","submitted_at":"2025-09-11T12:32:08Z","title":"DiTReducio: A Training-Free Acceleration for DiT-Based TTS via Progressive Calibration","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-04T19:11:59.490276Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2509.09748"},"observation_digest":"sha256:1e621e81b0d710afa12bb086d17eea6f347364a533969b1fee0b3fd174b258d0","observation_id":"0b2b2830-f9b6-4d17-90a5-b4d3fc598206","resolution":{"observed_at":"2026-08-04T19:11:59.490276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2509.19883","last_updated":"2026-04-11T06:12:16Z","snapshot_observed_at":"2026-08-02T07:43:41.108992Z","submitted_at":"2025-09-24T08:34:19Z","title":"CoMelSinger: Discrete Token-Based Zero-Shot Singing Synthesis With Structured Melody Control and Guidance","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-18T14:34:15.220263Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2509.19883"},"observation_digest":"sha256:338fbdc608064b3805baee5b4101c8ea98aafe19b26b760bac1456b04f470d47","observation_id":"eaedd33e-6e22-4792-acae-86dc149fecf8","resolution":{"observed_at":"2026-05-18T14:36:28.820222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T11:29:30.396377Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.04593","last_updated":"2026-06-08T05:49:30Z","snapshot_observed_at":"2026-08-06T08:55:28.794308Z","submitted_at":"2025-10-06T08:47:38Z","title":"UniVoice: Unifying Autoregressive ASR and Flow-Matching based TTS with Large Language Models","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T11:29:30.396377Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2510.04593"},"observation_digest":"sha256:0e5399cae84d32d2b5253cecd773150f429024ec0697bebd31d3f95df2c921e0","observation_id":"0a66343d-5dab-48a9-871d-109ba2cbdaee","resolution":{"observed_at":"2026-08-04T11:29:30.396377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T11:07:39.647731Z","title":"Preprint, arXiv:2406.05370","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2510.06927","last_updated":"2026-05-25T16:15:03Z","snapshot_observed_at":"2026-08-06T09:42:01.216340Z","submitted_at":"2025-10-08T12:07:57Z","title":"Position: Towards Responsible Evaluation for Text-to-Speech","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T11:07:39.647731Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2510.06927"},"observation_digest":"sha256:db696fa8965ac4883eca1befcc30ff8d1dc5568db55b35a3c7af902efdb335eb","observation_id":"35c49ad4-9c77-458f-b191-99ac9d25e756","resolution":{"observed_at":"2026-08-04T11:07:39.647731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-03T06:36:31.871558Z","title":"Vall-e 2: Neural codec lan- guage models are human parity zero-shot text to speech synthesizers.arXiv preprint arXiv:2406.05370,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.22661","last_updated":"2026-05-27T18:10:43Z","snapshot_observed_at":"2026-08-05T03:41:49.406487Z","submitted_at":"2026-01-30T07:27:48Z","title":"Evaluating and Rewarding LALMs for Expressive Role-Play TTS via Mean Continuation Log-Probability","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-03T06:36:31.871558Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2601.22661"},"observation_digest":"sha256:1185cf4a1b848b2e9bff1fb7500cef716e376c31a1e1c4e69122aa08e1d7a56b","observation_id":"21fdd2b4-abf1-44b8-a7b8-baf9025ebf8e","resolution":{"observed_at":"2026-08-03T06:36:31.871558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2603.05373","last_updated":"2026-04-11T06:15:04Z","snapshot_observed_at":"2026-08-07T04:20:21.141625Z","submitted_at":"2026-03-05T16:59:26Z","title":"Hierarchical Decoding for Discrete Speech Synthesis with Multi-Resolution Spoof Detection","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T15:07:27.418534Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2603.05373"},"observation_digest":"sha256:bd7a4999c9eb05e70c838e5ef56c77c553832c70d03e76b93059fe447f6494f2","observation_id":"a0b2c8d6-4306-4cb9-89aa-27d3603f54a1","resolution":{"observed_at":"2026-05-15T15:10:06.050969Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2604.08363","last_updated":"2026-04-09T15:27:22Z","snapshot_observed_at":"2026-07-06T22:57:26.678728Z","submitted_at":"2026-04-09T15:27:22Z","title":"CapTalk: Unified Voice Design for Single-Utterance and Dialogue Speech Generation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T17:15:28.918204Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2604.08363"},"observation_digest":"sha256:a36709e7113497939b603ead138e20577a37ebb21e410a75c3e123abb69e7bff","observation_id":"0e7816cc-9440-42fb-8945-f208a6c3d830","resolution":{"observed_at":"2026-05-11T07:15:59.846355Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2604.10065","last_updated":"2026-04-11T07:07:08Z","snapshot_observed_at":"2026-07-06T22:58:47.599383Z","submitted_at":"2026-04-11T07:07:08Z","title":"ASPIRin: Action Space Projection for Interactivity-Optimized Reinforcement Learning in Full-Duplex Speech Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T17:05:45.214298Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2604.10065"},"observation_digest":"sha256:349e5d301a115d89f9583795d050c260c989b67cdb91fcc3199f29ce8a1fd13f","observation_id":"6e4a4aa1-4677-401c-a4c4-1a60ac95ac1f","resolution":{"observed_at":"2026-05-11T07:35:58.821060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2604.11103","last_updated":"2026-04-24T05:22:49Z","snapshot_observed_at":"2026-07-06T22:59:36.571641Z","submitted_at":"2026-04-13T07:20:20Z","title":"ActorMind: Emulating Human Actor Reasoning for Speech Role-Playing","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T16:03:15.572657Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2604.11103"},"observation_digest":"sha256:4b316e7b08c270e577ccfd501faf955a0632f7a27c7fa2eb8aedf4ebd3068670","observation_id":"7f2bb8ca-5d3d-4fda-a860-ad6d9be317d6","resolution":{"observed_at":"2026-05-11T09:21:02.392402Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2604.16056","last_updated":"2026-08-03T16:51:20Z","snapshot_observed_at":"2026-08-06T23:37:42.663981Z","submitted_at":"2026-04-17T13:30:59Z","title":"AST: Adaptive, Seamless, and Training-Free Precise Speech Editing","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T07:59:34.622096Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2604.16056"},"observation_digest":"sha256:ce5feb9d460ba13f5010634504f3f24e3d54e7d415c793c1f7b1b955d53a56a6","observation_id":"3bb668d8-ef79-4305-acb6-567cefcafa54","resolution":{"observed_at":"2026-05-10T08:02:25.229581Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T05:29:37.288120Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers.arXiv preprint arXiv:2406.05370, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.16056","last_updated":"2026-08-03T16:51:20Z","snapshot_observed_at":"2026-08-06T23:37:42.663981Z","submitted_at":"2026-04-17T13:30:59Z","title":"AST: Adaptive, Seamless, and Training-Free Precise Speech Editing","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T05:29:37.288120Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2604.16056"},"observation_digest":"sha256:6d8291513f2933ca18d129a083a20e8e8be2cda4d9faecd5512c03f04ca1c793","observation_id":"85f745ae-e1e7-4840-ac6d-8a4fa94e34ae","resolution":{"observed_at":"2026-08-04T05:29:37.288120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2604.26296","last_updated":"2026-04-29T04:51:14Z","snapshot_observed_at":"2026-07-06T23:11:57.238808Z","submitted_at":"2026-04-29T04:51:14Z","title":"SPG-Codec: Exploring the Role and Boundaries of Semantic Priors in Ultra-Low-Bitrate Neural Speech Coding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T12:44:08.364835Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2604.26296"},"observation_digest":"sha256:fec9c55edea6054a0df82ed282fcf1ba2fd5b11218280e088a4874decadfc943","observation_id":"45e50250-0aaa-4e0c-90dd-762047a61968","resolution":{"observed_at":"2026-05-12T09:11:25.373223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2605.25669","last_updated":"2026-05-25T10:19:16Z","snapshot_observed_at":"2026-08-03T21:18:19.659327Z","submitted_at":"2026-05-25T10:19:16Z","title":"Ultra-Low-Bitrate Mel-Spectrogram-based Neural Speech Coding with Flow-Matching-based Refinement and Vocoding-driven Reconstruction","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-29T19:45:01.705167Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2605.25669"},"observation_digest":"sha256:fa7abadbde6e2239c4b1224481077bdc9b604d5d3449093356c0fa66265f4e49","observation_id":"195876d7-3f1f-44ab-8cad-3a89abb7c570","resolution":{"observed_at":"2026-06-29T19:53:55.892033Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2605.26136","last_updated":"2026-05-21T21:22:44Z","snapshot_observed_at":"2026-07-06T23:36:01.564747Z","submitted_at":"2026-05-21T21:22:44Z","title":"Eroding Trust in Real Speech: A Large-Scale Study of Human Audio Deepfake Perception","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T15:40:36.466069Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2605.26136"},"observation_digest":"sha256:434d6a6fe9594904321790494bea41cb9bf75582ebd524fbf507ae01cc87ed76","observation_id":"ba4299f2-3240-4434-810f-20192360e422","resolution":{"observed_at":"2026-06-30T15:44:48.374423Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2605.26672","last_updated":"2026-06-17T13:28:50Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:11:27Z","title":"Can We Hear from Events? Generating Speech from Event Camera","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T14:40:29.819278Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2605.26672"},"observation_digest":"sha256:740adee9a1e555333e3366730c2b02c5cbe86dde5dfd2312727749fd1d1c8dbe","observation_id":"77c95389-f79c-4cd9-95ca-df1846225053","resolution":{"observed_at":"2026-06-29T14:43:30.648577Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2605.29859","last_updated":"2026-05-28T12:39:36Z","snapshot_observed_at":"2026-08-01T16:23:26.516252Z","submitted_at":"2026-05-28T12:39:36Z","title":"MELD: Mel-Spectrogram-Based Speech Language Modeling with Discrete Latent Variables","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T05:34:55.822901Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2605.29859"},"observation_digest":"sha256:4bc07e13c48466161925223cd9916519fdcfe94e00277602d0f89f6c16ac4e8c","observation_id":"e3030e54-10c4-45dd-9755-9e24c05f2cd1","resolution":{"observed_at":"2026-06-29T05:43:08.711740Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.01677","last_updated":"2026-06-01T04:35:28Z","snapshot_observed_at":"2026-08-01T15:52:02.822088Z","submitted_at":"2026-06-01T04:35:28Z","title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-06-28T13:17:13.510587Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.01677"},"observation_digest":"sha256:6a9d156200e87b54da3f4940c68681efaf08ba581e7ebae7d4bfd45382482f5a","observation_id":"5b24562f-9744-4546-9833-a25226c5359b","resolution":{"observed_at":"2026-07-02T00:46:24.605010Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.03455","last_updated":"2026-06-02T10:33:20Z","snapshot_observed_at":"2026-08-05T18:48:21.902772Z","submitted_at":"2026-06-02T10:33:20Z","title":"WavTTS: Towards High-Quality Zero-Shot TTS via Direct Raw Waveform Modeling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T08:18:42.002083Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.03455"},"observation_digest":"sha256:b7b3f8e57756491a554ed228bfa0be335aa08678720464dcc6eaf4144b5b82e5","observation_id":"16dedabf-4f7f-4a58-8162-5d6a50496af0","resolution":{"observed_at":"2026-07-02T05:16:39.798400Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.05852","last_updated":"2026-06-04T08:27:17Z","snapshot_observed_at":"2026-08-04T06:57:18.394840Z","submitted_at":"2026-06-04T08:27:17Z","title":"UniVoice: A Unified Model for Speech and Singing Voice Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T23:56:14.198308Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.05852"},"observation_digest":"sha256:078f3a53c595e385ce5c31279577e1bf4d564077eded78f48a88dc68ba62e399","observation_id":"9350765f-4c89-45dc-affd-64f9f4f2041b","resolution":{"observed_at":"2026-07-02T15:17:08.326755Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.09019","last_updated":"2026-06-08T04:32:08Z","snapshot_observed_at":"2026-07-06T23:48:26.076260Z","submitted_at":"2026-06-08T04:32:08Z","title":"TLDR: Compressing Audio Tokens for Efficient Autoregressive Text-to-Speech","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T15:27:01.142747Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.09019"},"observation_digest":"sha256:966a9a8b82f1af2969ddcf49911b3c5c583042b8a6b51f1a66c0608829cdec95","observation_id":"298ce71c-92cb-475a-911c-c4223f441028","resolution":{"observed_at":"2026-07-03T03:07:35.864020Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.18485","last_updated":"2026-06-16T20:58:26Z","snapshot_observed_at":"2026-08-02T09:07:10.325810Z","submitted_at":"2026-06-16T20:58:26Z","title":"MagpieTTS-LF: Inference-Time Long-Form Speech Generation Without Training on Long-Form data","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T22:29:05.130421Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.18485"},"observation_digest":"sha256:3c96fecf49ff726742e3533b510ad02e507b4852bc0d185f1b9c527b2a103695","observation_id":"054d7587-0a6b-4ae9-971a-4feb9c37c6a1","resolution":{"observed_at":"2026-07-03T23:19:03.865644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.19823","last_updated":"2026-06-18T05:55:58Z","snapshot_observed_at":"2026-08-08T05:40:05.379801Z","submitted_at":"2026-06-18T05:55:58Z","title":"Low-Burden Data Augmentation for Dysarthric ASR via Zero-Shot Voice Cloning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-26T16:05:40.102418Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.19823"},"observation_digest":"sha256:7acba94872ecf972531379c80b5e02ae51edcad2e95e0b9e0629310a63408697","observation_id":"ca50811b-9982-4aba-a2e4-56b9bda8a131","resolution":{"observed_at":"2026-07-04T05:19:35.346427Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.20137","last_updated":"2026-06-18T12:00:24Z","snapshot_observed_at":"2026-08-06T03:59:56.653878Z","submitted_at":"2026-06-18T12:00:24Z","title":"PASQA: Pitch-Accent-Focused Speech Quality Assessment Model Trained on Synthetic Speech with Accent Errors","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-26T15:43:35.759550Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.20137"},"observation_digest":"sha256:fb41d13dd1a271c05468ebdb88c7aa9e2147dd01d3f43af3aaca398c304aa446","observation_id":"a8b77bdb-d009-464a-840b-7cd5be7d4fbb","resolution":{"observed_at":"2026-07-04T05:39:40.419205Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.20266","last_updated":"2026-06-18T14:14:14Z","snapshot_observed_at":"2026-08-07T16:53:00.938422Z","submitted_at":"2026-06-18T14:14:14Z","title":"Transcript-Free Flow-Matching Text-to-Speech via Speech Feature Conditioning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-26T15:38:13.819676Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.20266"},"observation_digest":"sha256:bb7750b58cad6769e95975cfef86b2ac756b7b5eded3982721d3a16a9236a0c8","observation_id":"ccf1f974-c7e1-4f05-bc22-628c2e7525fc","resolution":{"observed_at":"2026-07-04T05:39:40.774406Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2606.25403","last_updated":"2026-06-24T05:02:36Z","snapshot_observed_at":"2026-08-01T01:01:18.146622Z","submitted_at":"2026-06-24T05:02:36Z","title":"CrossAccent-TTS: Cross-Lingual Accent-Intensity Controllable Text-to-Speech via Disentangled Speaker and Accent Representations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-25T20:21:47.681999Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2606.25403"},"observation_digest":"sha256:e00c96365cabdb4c043f969569aeb9691ee571a45b7dc76c78ccd66e01901595","observation_id":"b86b6ea6-9515-4274-bc64-37edf17ea2f8","resolution":{"observed_at":"2026-07-04T20:20:07.163095Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2607.00387","last_updated":"2026-07-01T03:32:08Z","snapshot_observed_at":"2026-08-01T10:46:33.437238Z","submitted_at":"2026-07-01T03:32:08Z","title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","version":1},"reference_index":116,"source":"pdf_text","source_observed_at":"2026-07-02T05:52:55.818877Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.00387"},"observation_digest":"sha256:f338162d15e112d66321a1a0e1f4a598c8fa0fe978e92c149567fbc10a4b04dd","observation_id":"1c9e99ca-6521-446e-bb85-08545df4436f","resolution":{"observed_at":"2026-07-02T05:56:39.862333Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":"2406.05370","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-08T00:04:22.411558Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":"cs.CL","work_id":"5de8849e-7ff0-45e8-ba28-c43c1dbe805a","year":2024},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-07-11T07:46:45.098798Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":1},"reference_index":236,"source":"arxiv_source","source_observed_at":"2026-07-07T23:59:38.702609Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:3847c8e44b2ede5f648d35a5c50cc8bee77febf0c7bb8eaf16f277beaa538a24","observation_id":"48ea64e8-2bb3-48f4-8f5a-65b0cecc7052","resolution":{"observed_at":"2026-07-08T00:04:22.412990Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-11T07:46:49.059192Z","title":"arXiv preprint arXiv:2406.05370 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.05196","last_updated":"2026-07-07T15:36:48Z","snapshot_observed_at":"2026-07-11T07:46:45.098798Z","submitted_at":"2026-07-06T15:11:57Z","title":"Unified Audio Intelligence Without Regressing on Text Intelligence","version":2},"reference_index":236,"source":"arxiv_source","source_observed_at":"2026-07-11T07:46:49.059192Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.05196"},"observation_digest":"sha256:a1c0dbd63a1fc2ae3fd091a8e81fb1399563feb753f9d7c121b635c65a074864","observation_id":"a256841e-f3a6-48a0-8633-cfba7b2c8117","resolution":{"observed_at":"2026-07-11T07:46:49.059192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-01T17:37:38.794755Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17592","last_updated":"2026-07-20T06:20:17Z","snapshot_observed_at":"2026-08-07T11:02:21.659840Z","submitted_at":"2026-07-20T06:20:17Z","title":"SSTMark: Robust Training-Free Semantic-Level Speech Watermarking","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T17:37:38.794755Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.17592"},"observation_digest":"sha256:db6831fffbf6bc9869cdda1f7474b4d5f838829999eb3bc82d7a0f32a51f6b1f","observation_id":"17ef13f6-c1e3-4fa9-bdf7-dc79538eb0b0","resolution":{"observed_at":"2026-08-01T17:37:38.794755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-01T03:52:17.151139Z","title":"V ALL-E 2: Neural codec language models are human parity zero-shot text to speech synthesizers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23027","last_updated":"2026-07-25T03:59:26Z","snapshot_observed_at":"2026-08-07T02:28:24.488683Z","submitted_at":"2026-07-25T03:59:26Z","title":"Singlish, Can or Not? Fine-Tuning and Evaluating Zero-Shot TTS for Singapore English","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T03:52:17.151139Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.23027"},"observation_digest":"sha256:206c956bb93cc0538937fc1b2387a48bf23c13bdc5988a0703f3ecb8044838ae","observation_id":"41cc9bbd-15a6-4a9d-b01c-e1eafaa25407","resolution":{"observed_at":"2026-08-01T03:52:17.151139Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-07-31T15:08:47.779386Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24430","last_updated":"2026-07-27T13:42:37Z","snapshot_observed_at":"2026-08-06T19:11:09.358489Z","submitted_at":"2026-07-27T13:42:37Z","title":"Let Me Look at You: Advanced Facial Expression Modeling for Conversational Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-31T15:08:47.779386Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.24430"},"observation_digest":"sha256:de49835259fbc729d3a37464e7e16a69ab0da3a09d4f2b4a3447ca7a50dff580","observation_id":"ce2fba8b-b260-40ef-a172-decdda9b2d4c","resolution":{"observed_at":"2026-07-31T15:08:47.779386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-03T08:35:47.644694Z","title":"Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers.arXiv preprint arXiv:2406.05370, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29363","last_updated":"2026-07-31T12:51:24Z","snapshot_observed_at":"2026-08-05T23:15:08.967026Z","submitted_at":"2026-07-31T12:51:24Z","title":"Stable Autoregressive Speech Generation with Low-Frame-Rate High-Dimensional Continuous Tokens","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T08:35:47.644694Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2607.29363"},"observation_digest":"sha256:9dc8437629a0d985d9bc67b78ecd6d9d9aabc02f3a24cceeee7f6ae8de5e3175","observation_id":"7878a799-da21-4452-a938-411c8de1e7f4","resolution":{"observed_at":"2026-08-03T08:35:47.644694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-04T02:47:35.142024Z","title":"V ALL-E 2: Neural codec language models are human parity zero-shot text to speech syn- thesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00011","last_updated":"2026-06-19T17:35:42Z","snapshot_observed_at":"2026-08-07T13:04:54.841823Z","submitted_at":"2026-06-19T17:35:42Z","title":"DLLM-TTS: Block Discrete Diffusion Language Model for Text-to-Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T02:47:35.142024Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2608.00011"},"observation_digest":"sha256:29b22f4d75f2051a0a9ad2eca9ca3a5887307e38e8783737a17e3fccb889d581","observation_id":"3ee3d8f0-ae27-4ffb-a62f-2f41c2b2fb4c","resolution":{"observed_at":"2026-08-04T02:47:35.142024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.05370/citation-record","integrity":"/paper/2406.05370/integrity","json":"/paper/2406.05370/citation-record.json","paper":"/paper/2406.05370"},"outbound":[],"paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T08:54:58.638188Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 52 inbound Pith citation observations for arXiv:2406.05370."}