{"as_of":"2026-08-18T07:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:51eafb89151fd62dbdde7a017d785964935c5652e395c04d3ea192a18a5979ac","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:10:40.920937Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:10:40.788931Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-12T00:10:41.540631Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2608.08362","doi":null,"metadata_source":"pith","pith_arxiv_id":"2608.08362","snapshot_observed_at":"2026-08-12T00:10:41.540631Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","venue":"eess.AS","work_id":"12724a0a-8e91-413c-9efa-8d8047fb5591","year":2026},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.788931Z"},"links":{"cited_paper":"/paper/2608.08362","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:718f5ff16573101c8d7eba6771af014359c461a65760f74e3b83438e74c75cb3","observation_id":"6a494d1b-0a14-4601-98f0-8bc0bcb2ac01","resolution":{"observed_at":"2026-08-12T00:10:41.544806Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2608.08362/citation-record","integrity":"/paper/2608.08362/integrity","json":"/paper/2608.08362/citation-record.json","paper":"/paper/2608.08362"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.890533Z","title":null,"venue":null,"work_id":"dceb40f9-2013-4987-a404-6f384f238d82","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.774360Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:dd160a21c87d56dfda027d7175b328bf0da5c4ed3562895d639a29bba59df714","observation_id":"d3906908-4e24-4528-b219-2ec9d33f18fe","resolution":{"observed_at":"2026-08-12T00:10:41.893849Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.881517Z","title":null,"venue":null,"work_id":"15fa2a23-db14-42cf-a862-2402bd292ee6","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.778165Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:dbcaf31a18f7d0a6382e5ba9a5e82ce2919c138a0c5296e61dc7b692c27f7cbf","observation_id":"7ca53c06-e6f2-47ae-b9b7-64d9808f0602","resolution":{"observed_at":"2026-08-12T00:10:41.884413Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.872463Z","title":null,"venue":null,"work_id":"de28cee5-c532-472d-8572-20ac20276af3","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.781218Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:ba32922beb012ae1a3919b9f5988b22d23511ac8df4e97ceb0567ccb73082658","observation_id":"f3eb38eb-8e72-44bd-ae26-ebe1c70f1d3e","resolution":{"observed_at":"2026-08-12T00:10:41.875334Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.863365Z","title":null,"venue":null,"work_id":"313337e6-ce62-44e4-8fec-5d9e385a100a","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.785035Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:76318dd25e10ed1bcb516a36811381d24bedbe7fc5089f60b16efc5e5977b6e9","observation_id":"440044ac-ccbf-4749-bcf6-10ac8c78b8d1","resolution":{"observed_at":"2026-08-12T00:10:41.866574Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2608.08362","doi":null,"metadata_source":"pith","pith_arxiv_id":"2608.08362","snapshot_observed_at":"2026-08-12T00:10:41.540631Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","venue":"eess.AS","work_id":"12724a0a-8e91-413c-9efa-8d8047fb5591","year":2026},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.788931Z"},"links":{"cited_paper":"/paper/2608.08362","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:718f5ff16573101c8d7eba6771af014359c461a65760f74e3b83438e74c75cb3","observation_id":"6a494d1b-0a14-4601-98f0-8bc0bcb2ac01","resolution":{"observed_at":"2026-08-12T00:10:41.544806Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.853474Z","title":"outpainting","venue":null,"work_id":"b2810da6-d55d-491d-a7a4-78b16cac4362","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.792815Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:43e08571f92066d2032e24243fa23e70b503cd97c7bcba55c101b524e22f5805","observation_id":"33dac800-e9b4-4dcf-be69-a7637841b420","resolution":{"observed_at":"2026-08-12T00:10:41.856686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.843562Z","title":null,"venue":null,"work_id":"18005b19-abb8-4105-9ae3-1c95cc9d6176","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.796541Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:d5f0625670175c92da3ffb60c2dc8b20d3daad2585eeb0b30210bc1045bbde11","observation_id":"a0d2cb77-2aa9-4380-bf83-f557d79ce680","resolution":{"observed_at":"2026-08-12T00:10:41.846877Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.834220Z","title":null,"venue":null,"work_id":"c54f7e71-0ab8-4968-9769-4b0fb57db572","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.800135Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:047eb01d2994e1d22c3d9e574d72bc544fb1fb5bd35a7f705e938a539b2c275a","observation_id":"097e6d95-3fde-4a8f-96ae-2401dbab8ce6","resolution":{"observed_at":"2026-08-12T00:10:41.837289Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.824971Z","title":null,"venue":null,"work_id":"f830119e-26a5-49da-ad3e-74b281d03343","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.803689Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:cb087ca956e88a669b0de60ab7bbe7a21133edc4809babccdfb9bf9b311f1361","observation_id":"39e83fd4-f028-4ba0-a7c9-4a397bfa4e5b","resolution":{"observed_at":"2026-08-12T00:10:41.827861Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.814489Z","title":"First, our experiments are mainly conducted on English speech, so the effectiveness of CTRLSPEECHfor multilingual or code-switching synthesis re- mains unexplored","venue":null,"work_id":"b1e33f64-d814-4f92-aee9-7112e6b88ee9","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.806688Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:cc41b70fcfca68c56ee007929680a58750a9c3047cb2c90c392f32febfd92d9a","observation_id":"caf97688-a2fb-4f82-b633-172179f6c8d7","resolution":{"observed_at":"2026-08-12T00:10:41.818313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.803546Z","title":"2D- 16003984 through the Amazon-UT Austin HUB","venue":null,"work_id":"5d71e3be-3962-4e6f-b77f-37cdee8ae750","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.809810Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:1d1a1848be7c52bc08dea8b376e262ee582b77e49d0aeabf1b4d1e2742ee2ed4","observation_id":"92c5ee93-08e0-4bad-8be0-32c56eeae2c5","resolution":{"observed_at":"2026-08-12T00:10:41.807232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.793280Z","title":"All authors remain fully responsible for the con- tent of this manuscript","venue":null,"work_id":"33120f40-c729-4f12-b2d2-9dd3fbcd45ed","year":null},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.812951Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:3c1c5509148d4b101c2c5f74610e83d095cb4ee35ae41dd0b6cc5b8202bf25df","observation_id":"ee42a50d-1072-413f-9fe8-6934b39bcb0b","resolution":{"observed_at":"2026-08-12T00:10:41.796694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.816253Z","title":"Ditar: Diffusion transformer autoregressive modeling for speech generation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.816253Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:d318eaffa84811e11a0e29076312f630dcfeb6a27d617248f8cbabc150fc2d9d","observation_id":"a7d6d96c-9730-46e6-bc04-c727ef0716c1","resolution":{"observed_at":"2026-08-12T00:10:40.816253Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-12T00:10:40.819409Z","title":"Neural codec language mod- els are zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.819409Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:d9137b25d56ac35513e65ce0da6a7ec2b8c860831b63c90fc6eef03746e6406a","observation_id":"027dec4c-984b-4e54-97f3-e3d7d95a6796","resolution":{"observed_at":"2026-08-12T00:10:40.819409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03100","last_updated":"2024-04-23T08:38:03Z","snapshot_observed_at":"2026-08-16T14:12:35.270314Z","submitted_at":"2024-03-05T16:35:25Z","title":"NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03100","snapshot_observed_at":"2026-08-12T00:10:40.823347Z","title":"Naturalspeech 3: Zero-shot speech syn- thesis with factorized codec and diffusion models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.823347Z"},"links":{"cited_paper":"/paper/2403.03100","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:22ee8af17dfd5a442103be20c9063328a4839431b08caa14ddc43a77f4ea6b93","observation_id":"24ffd9c1-4804-4199-b409-a07fe33fd398","resolution":{"observed_at":"2026-08-12T00:10:40.823347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00750","last_updated":"2024-10-20T14:25:49Z","snapshot_observed_at":"2026-08-16T13:22:17.818052Z","submitted_at":"2024-09-01T15:26:30Z","title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.00750","snapshot_observed_at":"2026-08-12T00:10:40.826976Z","title":"Maskgct: Zero-shot text-to- speech with masked generative codec transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.826976Z"},"links":{"cited_paper":"/paper/2409.00750","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:2a312fa476238feef24858ee326e15b448817ba575aaa8739db376345eaad4d8","observation_id":"4dc0ab8f-49f9-4847-84ec-fe1532ef48a4","resolution":{"observed_at":"2026-08-12T00:10:40.826976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.830784Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.830784Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:0bc732d4994b39a6422eafb0bf2e34f32a55cf55ff80f0179ab602e509ccb5cc","observation_id":"6409aaac-722e-4444-bd63-62c7de4cd51a","resolution":{"observed_at":"2026-08-12T00:10:40.830784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.777737Z","title":"V oicecraft-x: Unifying multilingual, voice- cloning speech synthesis and speech editing,","venue":null,"work_id":"9a468f0d-10e0-4292-bab9-50e6e5cbf8ee","year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.834489Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:0bab85ab1167b16d13036f980433c82d5e733c68321a0a376517ddb635f2138e","observation_id":"1c7967d7-8218-47f1-a511-30027b4a84f4","resolution":{"observed_at":"2026-08-12T00:10:41.781590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.838301Z","title":"Prompttts: Control- lable text-to-speech with text descriptions,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.838301Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:116d1a8bcb1f9387dd930f6c0f012a89d4f6b472f3051de277bd4f6b7397cb64","observation_id":"c025b68e-de5c-4b9f-a609-f03798ad1bf6","resolution":{"observed_at":"2026-08-12T00:10:40.838301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.02285","last_updated":"2023-10-12T03:05:36Z","snapshot_observed_at":"2026-08-16T15:03:09.770881Z","submitted_at":"2023-09-05T14:45:27Z","title":"PromptTTS 2: Describing and Generating Voices with Text Prompt","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.02285","snapshot_observed_at":"2026-08-12T00:10:40.841801Z","title":"Prompttts 2: Describing and generating voices with text prompt,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.841801Z"},"links":{"cited_paper":"/paper/2309.02285","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:6c70c1078eab458f56281fb332ca4aced2d03e01e6c7e93b13c0de042cd91094","observation_id":"4a09c230-013b-4930-8825-116294c725d3","resolution":{"observed_at":"2026-08-12T00:10:40.841801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.762118Z","title":"Instructtts: Modelling expressive tts in discrete latent space with natural lan- guage style prompt,","venue":null,"work_id":"1eaeafb5-a9c6-4456-9e2b-1507c76b56ba","year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.846245Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:8d707780fa1666d7bf19d9aa40a8d4ed839485e24ae8319cf8d8b6a7ccd2161d","observation_id":"084a1113-77e1-4aaf-836c-35857e15c28e","resolution":{"observed_at":"2026-08-12T00:10:41.765506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-08-16T14:21:59.601701Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-12T00:10:40.849652Z","title":"Natural language guidance of high- fidelity text-to-speech with synthetic annotations,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.849652Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:4fd6cf19282c4732b55d8b1d74c0f3e1b27b4dd49d76c556def429ca0a103dbc","observation_id":"b6a76d86-87f8-48ce-ac8e-37bd5dc32208","resolution":{"observed_at":"2026-08-12T00:10:40.849652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.751812Z","title":"V oxinstruct: Expressive human instruction-to-speech generation with unified multilingual codec language modelling,","venue":null,"work_id":"93fc14bb-3ec6-4b89-9426-406b85eac696","year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.853557Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:901ce629eb9153c63430aad7b119278fd0c723e2085736284e51b2481eb78f10","observation_id":"02056422-fa70-46ed-b4bb-963fc2b1fe60","resolution":{"observed_at":"2026-08-12T00:10:41.755234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.856918Z","title":"Emovoice: Llm-based emotional text-to-speech model with freestyle text prompting,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.856918Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:182d79f96ca07bcfcaa89eb9719f3cea260c0dc2f70badcbe578c25212e23773","observation_id":"c700938b-0062-403e-a8ce-f40cb29a6b4c","resolution":{"observed_at":"2026-08-12T00:10:40.856918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.735802Z","title":"Scaling rich style- prompted text-to-speech datasets,","venue":null,"work_id":"39104a6f-a407-417a-a6bb-0d02d504e22d","year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.859754Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:297c2c38f9deef21d4cf75097b6438e107b6aa69dba49c81d8b7200f78609724","observation_id":"4e951b5e-af90-44a8-bb98-d41fec64bdc4","resolution":{"observed_at":"2026-08-12T00:10:41.739353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17589","last_updated":"2025-05-27T07:48:34Z","snapshot_observed_at":"2026-08-16T06:12:24.457686Z","submitted_at":"2025-05-23T07:55:21Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17589","snapshot_observed_at":"2026-08-12T00:10:40.862468Z","title":"Cosyvoice 3: Towards in-the- wild speech generation via scaling-up and post-training,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.862468Z"},"links":{"cited_paper":"/paper/2505.17589","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:e341f505c85c6d72b0cf9c23eea1b955d33fd96bb3bb31107ee45fc14455fb33","observation_id":"79554df0-aee5-4816-9128-5e0f94e067c7","resolution":{"observed_at":"2026-08-12T00:10:40.862468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.865457Z","title":"Vevo2: A unified and controllable frame- work for speech and singing voice generation,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.865457Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:b603ce433b721c7875cd441ee1d46d09b287b1a35f3f80450d1b4868d3cb127d","observation_id":"fe29e582-b9c6-49c5-be71-2ed5e983fda7","resolution":{"observed_at":"2026-08-12T00:10:40.865457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.14784","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.319759Z","title":"Mela-tts: Joint transformer-diffusion model with representation alignment for speech synthesis,","venue":null,"work_id":"437715fc-b95b-4497-81aa-e23828e4a0cc","year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.868011Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:2b9cf9ffa31c902b70d009b4642d858e6e58ec247624606c6fe907bdcfc101f9","observation_id":"77015045-1b72-499d-b93f-68edfef23e56","resolution":{"observed_at":"2026-08-12T00:10:41.326818Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.724536Z","title":"Streammel: Real-time zero-shot text-to-speech via interleaved continuous autoregressive model- ing,","venue":null,"work_id":"0a04080d-0a6c-4202-ac7e-caea72d36584","year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.871337Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:bae30f1421f39fe7ffdbbf3087e5ba5b8386afaf7958e35f73138acd2e6181fa","observation_id":"737ebbf5-97a5-4e61-babd-a06700e970b1","resolution":{"observed_at":"2026-08-12T00:10:41.728008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.873950Z","title":"High-fidelity audio compression with improved rvqgan,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.873950Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:7099fbe0f78ebbbca8454f2bb912757e927a22b3cf03a548f1f6cb20393ce1b0","observation_id":"da98a471-810b-4dd6-bc07-a3d615f58348","resolution":{"observed_at":"2026-08-12T00:10:40.873950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04658","last_updated":"2023-02-16T18:48:56Z","snapshot_observed_at":"2026-08-16T16:54:06.083594Z","submitted_at":"2022-06-09T17:56:10Z","title":"BigVGAN: A Universal Neural Vocoder with Large-Scale Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04658","snapshot_observed_at":"2026-08-12T00:10:40.876765Z","title":"Bigvgan: A universal neural vocoder with large-scale training,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.876765Z"},"links":{"cited_paper":"/paper/2206.04658","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:d6b01bde6e90edca58eb988b34ccb1cce95ecf9927f404eabf0f3b737b491f85","observation_id":"5ff38436-b674-4f67-bf61-c21289a03c4c","resolution":{"observed_at":"2026-08-12T00:10:40.876765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-12T00:10:40.880326Z","title":"Cosyvoice: A scalable multi- lingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.880326Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:bd73ad2c050308cc2f729f49f9591acb8e3ea72843faa5b35b6a4440291ce4ca","observation_id":"beed7ec7-66d0-4a8d-bd5d-29038c9034eb","resolution":{"observed_at":"2026-08-12T00:10:40.880326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-16T06:25:22.037199Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-12T00:10:40.883298Z","title":"Cosyvoice 2: Scalable stream- ing speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.883298Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:4fd1351c2c03064e9bfa0cfa46d46d2ca4bc760b11f5a40cbc4a2fa277288c98","observation_id":"0755047a-85de-4b8a-bc6b-0994e4d246ff","resolution":{"observed_at":"2026-08-12T00:10:40.883298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.601539Z","title":"World: a vocoder-based high-quality speech synthesis system for real-time applications,","venue":null,"work_id":"771e6e1f-b70c-4cd0-ab1b-6ce07856ca51","year":2016},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.886358Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:d9e86f5686c0d2b20d4e82150b204d6066a9ff7c74e463a690959485b951ed97","observation_id":"c6ff20da-d32e-4471-a84a-e20044149be6","resolution":{"observed_at":"2026-08-12T00:10:41.605365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.590505Z","title":"Fast and reliable f0 estimation method based on the period extraction of vocal fold vi- bration of singing voice and speech,","venue":null,"work_id":"a0439151-0885-4050-a39f-e8bb97abdbbb","year":2009},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.889367Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:9f7fba79755e4ab12c2ddb6af16e5683dfdef194d67b1cbe3bd55635bbf5657f","observation_id":"189bfeb3-c9ad-4c0e-8c7e-ef64b20f164f","resolution":{"observed_at":"2026-08-12T00:10:41.594371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.892028Z","title":"Continuous-token diffu- sion for speaker-referenced tts in multimodal llms,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.892028Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:6f62932ae8ea4c3d8d883263422a347ddbc2631dd976d477c5a045851843d880","observation_id":"b72ed2be-5ade-4a44-acc6-a08a36515fb1","resolution":{"observed_at":"2026-08-12T00:10:40.892028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.895164Z","title":"Emilia: An extensive, multilingual, and diverse speech dataset for large-scale speech generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.895164Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:789e5279cc79fd876042a15e8957c67ae27e792ba0943a48a696e588f9699010","observation_id":"c808d725-d472-4221-8b5e-224e44c2a1e5","resolution":{"observed_at":"2026-08-12T00:10:40.895164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.06909","last_updated":"2021-06-13T04:09:16Z","snapshot_observed_at":"2026-08-16T18:17:33.981607Z","submitted_at":"2021-06-13T04:09:16Z","title":"GigaSpeech: An Evolving, Multi-domain ASR Corpus with 10,000 Hours of Transcribed Audio","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.06909","snapshot_observed_at":"2026-08-12T00:10:40.898571Z","title":"Gigaspeech: An evolving, multi-domain asr corpus with 10,000 hours of transcribed audio,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.898571Z"},"links":{"cited_paper":"/paper/2106.06909","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:987417991a29fe52033f0571e46936b93913b41e193a4a32d428c113f6668a1a","observation_id":"d8b16eb8-10f0-4ab0-b52d-b78e3a58f35a","resolution":{"observed_at":"2026-08-12T00:10:40.898571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-08-15T11:55:32.600679Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-12T00:10:40.901548Z","title":"Seed-tts: A family of high-quality versatile speech generation models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.901548Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:a3adcebe8b4e6cc77a291d336844f3bc2634b3ea1e9d692fccaea75f6e6f35fb","observation_id":"ddd65d55-f795-4692-8566-41c299c3f5cd","resolution":{"observed_at":"2026-08-12T00:10:40.901548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.904905Z","title":"The lj speech dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.904905Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:5ddae4a237ceda92043e206a62d2ca935d756f2babcfa143b5c9152b778fe9ff","observation_id":"b8e17983-8473-41d0-9f8d-acb9088a8536","resolution":{"observed_at":"2026-08-12T00:10:40.904905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.907818Z","title":"Semantic-vae: Semantic- alignment latent representation for better speech synthesis,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.907818Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:c0872c2efb4ab7895f3dffbba4eb4cf90a576b16a5e5bc8ae1e08dbe764dd20f","observation_id":"5f85e8df-6a19-498c-bf75-d4205c79f7bd","resolution":{"observed_at":"2026-08-12T00:10:40.907818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.12598","last_updated":"2022-07-26T01:42:07Z","snapshot_observed_at":"2026-08-14T06:37:15.299690Z","submitted_at":"2022-07-26T01:42:07Z","title":"Classifier-Free Diffusion Guidance","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.12598","snapshot_observed_at":"2026-08-12T00:10:40.910596Z","title":"Classifier-free diffusion guidance,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.910596Z"},"links":{"cited_paper":"/paper/2207.12598","citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:b9dc268b56f8d936eacd238b5472c283d03ceb2b5f3f848752306bc3cdbadfab","observation_id":"da8fe183-b02b-44f7-a098-089abdda655d","resolution":{"observed_at":"2026-08-12T00:10:40.910596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.913810Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.913810Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:488b915ee655e6bdfb84ac279a8da3f70d448f4ddabe25f18e497451616ac05a","observation_id":"306d53bf-9ffc-46cf-bb8f-106d97ccf6a7","resolution":{"observed_at":"2026-08-12T00:10:40.913810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:40.917178Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.917178Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:b5b94160f8b19f6705e12a19519eedd2b2073d2975daf4bd0f6dd398ffd1eb7b","observation_id":"2d0016ae-58fc-4a76-b7b2-9bf848779e45","resolution":{"observed_at":"2026-08-12T00:10:40.917178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:10:41.551855Z","title":"Drawspeech: Expressive speech synthesis using prosodic sketches as control conditions,","venue":null,"work_id":"f2e55ff8-ff17-43ec-91a6-63eef731f772","year":2025},"citing_paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T00:10:40.920937Z"},"links":{"citing_paper":"/paper/2608.08362"},"observation_digest":"sha256:f6b430d1300a88c75f554e8950f7c023aa7dca040be757482af8455018ad138b","observation_id":"9ed6a26b-8b9c-4c3b-901a-7ed2ba1a11a2","resolution":{"observed_at":"2026-08-12T00:10:41.556446Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.08362","last_updated":"2026-08-08T23:17:09Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-18T06:45:08.583632Z","submitted_at":"2026-08-08T23:17:09Z","title":"CtrlSpeech: Coarse-to-Fine Control for Expressive Speech Synthesis"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":31,"verified_exact":1,"verified_fuzzy":12},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 1 inbound Pith citation observation for arXiv:2608.08362."}