{"as_of":"2026-08-07T17:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2e30a561b956f7cbabb811ba82198654a3ea58395e9c90f7b08bdf190de83b96","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:33:51.142199Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":5,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-07T06:02:40.568271Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-13T06:19:09.507440Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2412.10117"},"observation_digest":"sha256:c72edbbad90ed2ea13859b734f81076b0327b7cd067002238959bf814c1e2c06","observation_id":"381e5d4b-ee52-4d40-b293-c6eadfcfbea1","resolution":{"observed_at":"2026-05-13T06:19:09.598879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-07T15:33:51.142199Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14648","last_updated":"2025-05-20T17:36:41Z","snapshot_observed_at":"2026-08-07T15:28:11.502429Z","submitted_at":"2025-05-20T17:36:41Z","title":"Vox-Profile: A Speech Foundation Model Benchmark for Characterizing Diverse Speaker and Speech Traits","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:33:51.142199Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2505.14648"},"observation_digest":"sha256:c0828b22065ce7753833a78003637417612e053ccd24b816ef329affff0fefc6","observation_id":"1222f66f-e60a-4da6-a341-18a4d700dfa7","resolution":{"observed_at":"2026-08-07T15:33:51.142199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-07T14:32:42.812459Z","title":"Natural language guidance of high- fidelity text-to-speech with synthetic annotations,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18609","last_updated":"2025-05-27T14:19:23Z","snapshot_observed_at":"2026-08-07T14:26:42.682601Z","submitted_at":"2025-05-24T09:16:14Z","title":"RASMALAI: Resources for Adaptive Speech Modeling in Indian Languages with Accents and Intonations","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:32:42.812459Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2505.18609"},"observation_digest":"sha256:7c95a0c49a9a6feda4e3720e85953c264a379241be042cac6b8149bffa5b73c5","observation_id":"db7c9d04-f6fa-4db1-9bc1-def38a7a896a","resolution":{"observed_at":"2026-08-07T14:32:42.812459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-07T14:26:04.837891Z","title":"Natural language guidance of high- fidelity text-to-speech with synthetic annotations,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18972","last_updated":"2025-05-25T04:43:17Z","snapshot_observed_at":"2026-08-07T14:20:20.288742Z","submitted_at":"2025-05-25T04:43:17Z","title":"Revival with Voice: Multi-modal Controllable Text-to-Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:26:04.837891Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2505.18972"},"observation_digest":"sha256:2b383dc528405e5daa49e927ff4b030a831963d7cef8cde36d0423f91980ee71","observation_id":"1d643197-ab5f-49a1-96c9-e56bb18d5d6e","resolution":{"observed_at":"2026-08-07T14:26:04.837891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-07T14:19:55.191205Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.ArXiv, abs/2402.01912, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19462","last_updated":"2025-05-31T22:36:04Z","snapshot_observed_at":"2026-08-07T16:55:50.852560Z","submitted_at":"2025-05-26T03:35:44Z","title":"VoiceStar: Robust Zero-Shot Autoregressive TTS with Duration Control and Extrapolation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:55.191205Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2505.19462"},"observation_digest":"sha256:1831be914c003d0be60726ab76bea627898565e041e2b446467e500e3189a72c","observation_id":"a0230389-f536-413d-b453-4a4d8395e27d","resolution":{"observed_at":"2026-08-07T14:19:55.191205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T23:49:29.502416Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.16310","last_updated":"2025-06-19T13:35:05Z","snapshot_observed_at":"2026-08-06T23:41:50.460992Z","submitted_at":"2025-06-19T13:35:05Z","title":"Optimizing Multilingual Text-To-Speech with Accents & Emotions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T23:49:29.502416Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2506.16310"},"observation_digest":"sha256:765ae994e48cf8f32ddc4b6c94ee71e297eb51f1a6e3bc6104ec6a80a4e57cfb","observation_id":"de9f7812-7aa8-4c1b-ba5a-0257ba1e96db","resolution":{"observed_at":"2026-08-06T23:49:29.502416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T23:10:13.177749Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.19502","last_updated":"2025-07-15T06:04:25Z","snapshot_observed_at":"2026-08-07T17:10:15.118331Z","submitted_at":"2025-06-24T10:40:23Z","title":"MATE: LLM-Powered Multi-Agent Translation Environment for Accessibility Applications","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T23:10:13.177749Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2506.19502"},"observation_digest":"sha256:03451b4ed2d13b0c6bd242059f9c8912d272cce575e30dd0ff6868d2b24279e9","observation_id":"fb386c66-e20f-43ce-a4ca-d2c323e05703","resolution":{"observed_at":"2026-08-06T23:10:13.177749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T21:11:40.456914Z","title":"Natural language guidance of high- fidelity text-to-speech with synthetic annotations,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.00808","last_updated":"2025-07-02T05:42:47Z","snapshot_observed_at":"2026-08-06T21:04:23.906292Z","submitted_at":"2025-07-01T14:40:51Z","title":"Multi-interaction TTS toward professional recording reproduction","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:11:40.456914Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2507.00808"},"observation_digest":"sha256:73afbf5bda5214577dae3a5e3ddf5c4ac63e65d3e54bd315fb31e11639b85190","observation_id":"5c2cf6a8-f131-4464-984c-ec5e78ae53f9","resolution":{"observed_at":"2026-08-06T21:11:40.456914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T18:36:07.484115Z","title":"Lyth and S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07799","last_updated":"2025-07-10T14:26:28Z","snapshot_observed_at":"2026-08-06T18:29:45.274111Z","submitted_at":"2025-07-10T14:26:28Z","title":"SecureSpeech: Prompt-based Speaker and Content Protection","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T18:36:07.484115Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2507.07799"},"observation_digest":"sha256:958a83ea4e99b87a8d403569c4cbe7cdd0419b65ea1221adb0e1a32fdc82ac5c","observation_id":"9d1545d8-6f18-4bb2-92e6-f9bf768217e7","resolution":{"observed_at":"2026-08-06T18:36:07.484115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T18:21:33.855401Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-06T18:11:43.215379Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.855401Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:432021dfd1941ca80da33b4df0222b52ca61f34ab55856e2f29420f0563575b6","observation_id":"cdb68026-84dd-4f38-be9c-6b9913761e75","resolution":{"observed_at":"2026-08-06T18:21:33.855401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T14:50:04.214444Z","title":"ArXiv abs/2402.01912 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.17563","last_updated":"2025-07-23T14:53:50Z","snapshot_observed_at":"2026-08-07T03:22:50.485332Z","submitted_at":"2025-07-23T14:53:50Z","title":"BoSS: Beyond-Semantic Speech","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T14:50:04.214444Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2507.17563"},"observation_digest":"sha256:541e849d2eb07c865b1ba1504257f4c6f09af308b7228af61dc099fae2fd50f6","observation_id":"aea71d72-4db1-4369-9af7-bc31544e2993","resolution":{"observed_at":"2026-08-06T14:50:04.214444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T12:49:19.629984Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.21463","last_updated":"2025-07-29T03:06:31Z","snapshot_observed_at":"2026-08-06T12:49:15.030804Z","submitted_at":"2025-07-29T03:06:31Z","title":"SpeechFake: A Large-Scale Multilingual Speech Deepfake Dataset Incorporating Cutting-Edge Generation Methods","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T12:49:19.629984Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2507.21463"},"observation_digest":"sha256:5f79b7177ac8b4de64aaa1d7222dd83789687d7d52a18147b3ae97f6d733a6fc","observation_id":"1ce65cf1-7f38-4863-bf96-b730fdb28a2c","resolution":{"observed_at":"2026-08-06T12:49:19.629984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T20:04:48.453562Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.11326","last_updated":"2025-08-15T08:53:56Z","snapshot_observed_at":"2026-08-07T04:02:54.878172Z","submitted_at":"2025-08-15T08:53:56Z","title":"MoE-TTS: Enhancing Out-of-Domain Text Understanding for Description-based TTS via Mixture-of-Experts","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T20:04:48.453562Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2508.11326"},"observation_digest":"sha256:c1ceb1875282589eda890625bb8a07acca4097e8a6b0320afe8a50b512baddbc","observation_id":"604d43fb-5ff3-4590-b857-a3b1e3df899f","resolution":{"observed_at":"2026-08-05T20:04:48.453562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-04T18:58:04.179780Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09550","last_updated":"2025-09-12T06:43:25Z","snapshot_observed_at":"2026-08-07T14:59:36.196816Z","submitted_at":"2025-09-11T15:39:59Z","title":"Finite Scalar Quantization Enables Redundant and Transmission-Robust Neural Audio Compression at Low Bit-rates","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T18:58:04.179780Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2509.09550"},"observation_digest":"sha256:924c9ceb7416a8df4cec56619e3bd0b0bd89c0b43c8b8b67f3734fdca0a631b5","observation_id":"aeae0d3f-9346-482b-8379-4d91f7bfc692","resolution":{"observed_at":"2026-08-04T18:58:04.179780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2601.15621","last_updated":"2026-01-22T03:51:43Z","snapshot_observed_at":"2026-08-04T12:22:34.670840Z","submitted_at":"2026-01-22T03:51:43Z","title":"Qwen3-TTS Technical Report","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T19:24:56.057631Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2601.15621"},"observation_digest":"sha256:2d8d7c30afaed558c1e47d0909a866ff3b99a0f5090c95b905a4aabe2e08c185","observation_id":"160ea297-49c7-498b-8f39-1db404dc9e29","resolution":{"observed_at":"2026-05-16T19:24:56.136038Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2603.02364","last_updated":"2026-06-27T13:26:12Z","snapshot_observed_at":"2026-08-02T19:24:54.855385Z","submitted_at":"2026-03-02T20:11:06Z","title":"When Spoof Detectors Travel: Evaluation Across 66 Languages in the Low-Resource Language Spoofing Corpus","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-15T16:29:10.899659Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2603.02364"},"observation_digest":"sha256:c4d13ae72703311f6b4e9d10bee34a58ffe0c1d22561ecad720027910b5c81e7","observation_id":"949c6284-1e5f-489c-ab16-b3c6e92f45be","resolution":{"observed_at":"2026-05-15T16:30:09.664146Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-02T19:24:56.523051Z","title":"Natural Language Guidance of High- Fidelity Text-to-Speech with Synthetic Annotations,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.02364","last_updated":"2026-06-27T13:26:12Z","snapshot_observed_at":"2026-08-02T19:24:54.855385Z","submitted_at":"2026-03-02T20:11:06Z","title":"When Spoof Detectors Travel: Evaluation Across 66 Languages in the Low-Resource Language Spoofing Corpus","version":4},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T19:24:56.523051Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2603.02364"},"observation_digest":"sha256:fbf93498dc0b61df77c4f96a15479fc48a501b33dfe82500b13e95d168dd60a9","observation_id":"91288b67-7835-4021-acc9-f75b037dc95e","resolution":{"observed_at":"2026-08-02T19:24:56.523051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2604.17958","last_updated":"2026-04-20T08:39:55Z","snapshot_observed_at":"2026-07-06T23:04:55.190916Z","submitted_at":"2026-04-20T08:39:55Z","title":"MINT-Bench: A Comprehensive Multilingual Benchmark for Instruction-Following Text-to-Speech","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T04:03:38.919545Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2604.17958"},"observation_digest":"sha256:b18365972e1bcf952a50539fb9d6ef02ac27ba25675aa984a482a5552b11879c","observation_id":"5ff90eac-0152-49b9-a645-8644c62c3689","resolution":{"observed_at":"2026-05-10T04:04:47.244302Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2604.19330","last_updated":"2026-04-29T12:12:36Z","snapshot_observed_at":"2026-08-03T01:33:20.785741Z","submitted_at":"2026-04-21T10:58:48Z","title":"Text-To-Speech with Chain-of-Details: modeling temporal dynamics in speech generation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T01:01:06.094276Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2604.19330"},"observation_digest":"sha256:f9f3d8d4ee0e2d1e63b752a9a8d9b0556e63b6c893812abbc7237d75008285e0","observation_id":"4b13ef46-b70c-4ac7-87e6-6a8f1720f315","resolution":{"observed_at":"2026-05-10T01:04:50.107872Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2605.00861","last_updated":"2026-04-21T10:34:41Z","snapshot_observed_at":"2026-07-06T23:14:11.211164Z","submitted_at":"2026-04-21T10:34:41Z","title":"Voice Mapping of Text-to-Speech Systems: A Metric-Based Approach for Voice Quality Assessment","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T01:07:50.243903Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2605.00861"},"observation_digest":"sha256:2a123ecb7ef17ccdb05e4d1fc2d2691b63a60b38b0067f677b78f4107a616c32","observation_id":"8758a34a-8d0e-4f9b-b29b-0d694d21bec0","resolution":{"observed_at":"2026-05-10T01:10:09.324177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2605.12242","last_updated":"2026-05-12T15:11:36Z","snapshot_observed_at":"2026-07-06T23:23:59.123377Z","submitted_at":"2026-05-12T15:11:36Z","title":"Mind the Pause: Disfluency-Aware Objective Tuning for Multilingual Speech Correction with LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T05:20:35.333172Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2605.12242"},"observation_digest":"sha256:38ba6b7b871e96afe47c4a092d2ec443a2aeaca65a0d9dd3a9f64a2f19380460","observation_id":"12c7d0ed-e83f-4cc2-907b-800c1af311b4","resolution":{"observed_at":"2026-05-13T05:27:18.771185Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-07-13T00:18:36.390365Z","title":"InProceedings of the AAAI Conference on Artificial Intelligence, volume 38, pages 18698– 18706","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.27376","last_updated":"2026-04-09T01:06:26Z","snapshot_observed_at":"2026-07-13T00:18:29.539736Z","submitted_at":"2026-04-09T01:06:26Z","title":"Unlocking Fine-Grained and Within-Utterance Speaking Style Control in Prompt-Based Text-to-Speech Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-13T00:18:36.390365Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2605.27376"},"observation_digest":"sha256:d0e7a7833ff4dbad0334de7addf4b6523eb81f42162821185f11cbc94050fb35","observation_id":"b0b47817-d91b-4f24-b4e5-bb1081c9a2ee","resolution":{"observed_at":"2026-07-13T00:18:36.390365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2606.05889","last_updated":"2026-06-04T08:58:57Z","snapshot_observed_at":"2026-07-06T23:45:47.161248Z","submitted_at":"2026-06-04T08:58:57Z","title":"GLASS: GRPO-Trained LoRA for Acoustic Style Steering in Zero-Shot Text-to-Speech","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-27T23:53:55.385445Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2606.05889"},"observation_digest":"sha256:9d43998a1d932153890cd0aedd2c86101aaafb18843ec6e6c4327e394f9cf092","observation_id":"d9f6ae9c-89d1-4459-9a1f-92ba7f33e61a","resolution":{"observed_at":"2026-07-02T15:27:04.903956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2606.06928","last_updated":"2026-06-05T05:43:15Z","snapshot_observed_at":"2026-07-06T23:46:37.903443Z","submitted_at":"2026-06-05T05:43:15Z","title":"VoxCPM2 Technical Report","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T21:18:22.911332Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2606.06928"},"observation_digest":"sha256:2c0691b96d97a27f3743058199614726a91ee60169e242cbb6b30c9dceedd7eb","observation_id":"f00501ae-1501-41ad-b754-7230a5da2bd1","resolution":{"observed_at":"2026-07-02T19:47:19.727005Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2606.12199","last_updated":"2026-06-10T15:19:15Z","snapshot_observed_at":"2026-08-06T23:04:23.994908Z","submitted_at":"2026-06-10T15:19:15Z","title":"Which Speech Representation Better Matches Text-Native Reasoning? A Study of Speech-Text Alignment on Frame Rate and Representation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T08:18:23.182355Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2606.12199"},"observation_digest":"sha256:6bea4fd1fd5dbf81cfe85b9898f3f2af605f8fc2009413ae15f3524e8908dd28","observation_id":"b681c0fc-bc11-41d0-9b8e-f75c196e9e85","resolution":{"observed_at":"2026-07-03T13:18:12.790821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2606.19209","last_updated":"2026-06-18T16:42:05Z","snapshot_observed_at":"2026-08-05T18:10:28.030171Z","submitted_at":"2026-06-17T15:45:43Z","title":"FineCombo-TTS: Collaborative and Precise Controllable Speech Synthesis Using Text Descriptions and Reference Speech","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-26T19:09:33.605232Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2606.19209"},"observation_digest":"sha256:e946f63194952806e483bd8f8f0ac3683c829f3da0a723c2035cff8487a29e01","observation_id":"3ad237b5-80b3-46d7-9f7a-40afd9551ec8","resolution":{"observed_at":"2026-07-04T02:49:24.577822Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":"2402.01912","doi":"10.48550/arxiv.2402.01912","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912","venue":"arXiv (Cornell University)","work_id":"273353a0-dd82-4b8b-9507-22052ee19859","year":2024},"citing_paper":{"arxiv_id":"2606.20650","last_updated":"2026-06-08T06:54:55Z","snapshot_observed_at":"2026-08-02T23:08:38.292740Z","submitted_at":"2026-06-08T06:54:55Z","title":"EmoInstruct-TTS: Dual-Path Instruction-Guided Emotional Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T17:01:13.972071Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2606.20650"},"observation_digest":"sha256:b9c8bb17b590dcaa3a6cdfd52bf94e18a93dd76785b34e0ef02ee198221c603b","observation_id":"4655fea6-e4ec-4da8-8bdb-f9354cff3497","resolution":{"observed_at":"2026-07-03T00:47:30.552835Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-01T18:46:13.750265Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations.arXiv preprint arXiv:2402.01912, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18662","last_updated":"2026-07-19T11:20:25Z","snapshot_observed_at":"2026-08-06T18:14:48.659162Z","submitted_at":"2026-07-19T11:20:25Z","title":"Staged Depth-Pruning Distillation of a Flow-Matching Text-to-Speech Teacher: A Compact Hindi Speech Synthesizer","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T18:46:13.750265Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2607.18662"},"observation_digest":"sha256:57be105c6bc482ba967bc673c0d91a1a7ac5b3bf31d0628d7a707530d900cbba","observation_id":"7e2215f8-42e7-4a5b-9347-715b6a74b32e","resolution":{"observed_at":"2026-08-01T18:46:13.750265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-04T16:29:26.024843Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-07T17:19:23.197158Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-04T16:29:26.024843Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:d19cc16a898ada860930570e56533c2431b29bc39368d5645452318123949621","observation_id":"69f255a2-f5d5-4f26-a467-9c911d7b2c6a","resolution":{"observed_at":"2026-08-04T16:29:26.024843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-07T00:14:46.524756Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02023","last_updated":"2026-08-04T08:54:48Z","snapshot_observed_at":"2026-08-07T17:19:23.197158Z","submitted_at":"2026-08-03T10:20:52Z","title":"SwanTale: Unified Multi-Speaker Speech and Audio Generation for Instruct and Zero-Shot Tasks","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T00:14:46.524756Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2608.02023"},"observation_digest":"sha256:4d9c0e229965c4f56c7fdef87fb4a9126301c136e2fff7e5120f6604fa9806e7","observation_id":"d33c18b7-0caa-4bd3-9a9e-2e5011ebca0f","resolution":{"observed_at":"2026-08-07T00:14:46.524756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-04T10:58:58.620852Z","title":"Microsoft Corporation,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02235","last_updated":"2026-08-03T13:49:11Z","snapshot_observed_at":"2026-08-07T10:30:33.875274Z","submitted_at":"2026-08-03T13:49:11Z","title":"Domain-Specific Evaluation of Text-to-Speech Systems: A Multi-Metric Benchmarking Study","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T10:58:58.620852Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2608.02235"},"observation_digest":"sha256:001f6570725b010552e4e0fa9ad58e66b0ad9dd48d6f90aa1d039d1906228bd0","observation_id":"8851c956-d783-434b-9400-67f0ca5946d0","resolution":{"observed_at":"2026-08-04T10:58:58.620852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.01912/citation-record","integrity":"/paper/2402.01912/integrity","json":"/paper/2402.01912/citation-record.json","paper":"/paper/2402.01912"},"outbound":[],"paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-07-06T17:24:41.012207Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 31 inbound Pith citation observations for arXiv:2402.01912."}