{"as_of":"2026-08-07T21:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e970a25d1fd44db6187f3404b254e196dc6b916e3a32f810d09e527407483b10","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:21:11.541084Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-13T00:18:36.390365Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02742","snapshot_observed_at":"2026-07-13T00:18:36.390365Z","title":"Dake Guo, Xinfa Zhu, Liumeng Xue, Yongmao Zhang, Wenjie Tian, and Lei Xie","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.27376","last_updated":"2026-04-09T01:06:26Z","snapshot_observed_at":"2026-07-13T00:18:29.539736Z","submitted_at":"2026-04-09T01:06:26Z","title":"Unlocking Fine-Grained and Within-Utterance Speaking Style Control in Prompt-Based Text-to-Speech Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-13T00:18:36.390365Z"},"links":{"cited_paper":"/paper/2506.02742","citing_paper":"/paper/2605.27376"},"observation_digest":"sha256:065b5bc329b509c69ed2442354105d81fdfdd77d9c5c7c32d547ba4218bdbdd7","observation_id":"ab8d65e2-33a6-4319-b9f3-37bc6a94224c","resolution":{"observed_at":"2026-07-13T00:18:36.390365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02742/citation-record","integrity":"/paper/2506.02742/integrity","json":"/paper/2506.02742/citation-record.json","paper":"/paper/2506.02742"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.446538Z","title":"Text-to-speech synthesis based on latent variable conversion using diffusion probabilistic model and variational autoencoder,","venue":null,"work_id":"b4dc3907-bedb-43b2-a841-ade29e73cb89","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.227860Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:cb533eac03b853bc6f6b9bd41583ce319b36ea5a6b4b2fade32dfad59cc2b025","observation_id":"ecce2313-9455-4c17-a79e-d0cd65a7ac57","resolution":{"observed_at":"2026-08-07T11:21:14.449983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.435348Z","title":"A vector quantized approach for text to speech synthesis on real-world spontaneous speech,","venue":null,"work_id":"87cdafcf-c19e-419b-8fe4-05b023d6fd4c","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.283371Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:5316af03a7ca6e9cefbd0adba2eb23a5536d88d715943525243a3b3ef42804da","observation_id":"711828ad-1b2d-48ca-bf52-46f70d9d2622","resolution":{"observed_at":"2026-08-07T11:21:14.438992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.423412Z","title":"Text to speech synthesis: A systematic review, deep learning based architecture and future research direction,","venue":null,"work_id":"da83eca1-54a4-447b-aacb-0f6645de05d8","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.394854Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b2eba9ee7755a24a1af4130309d28cd19265656e789057bdc60086bbe7d04c7d","observation_id":"3c09f19e-818a-44e6-a4df-7c27647eb4c4","resolution":{"observed_at":"2026-08-07T11:21:14.427231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.412759Z","title":"A style control technique for hmm-based expressive speech synthesis,","venue":null,"work_id":"83fb7ce8-f8cc-43ac-a18b-c9fbf969cb40","year":2007},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.499770Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:5e3573f6337fffcdc3651e2f5c2c2e1e6ef4947c8bff8cd3dedc073dd5908b59","observation_id":"9ebae707-3b16-4da4-a1b0-e46f971d4c15","resolution":{"observed_at":"2026-08-07T11:21:14.416305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05187","last_updated":"2023-12-08T17:18:42Z","snapshot_observed_at":"2026-08-06T14:36:57.042438Z","submitted_at":"2023-12-08T17:18:42Z","title":"Seamless: Multilingual Expressive and Streaming Speech Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05187","snapshot_observed_at":"2026-08-07T11:21:08.605193Z","title":"Seamless: Multilingual expressive and streaming speech translation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.605193Z"},"links":{"cited_paper":"/paper/2312.05187","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:c4095e45388636172935c170ba5d1122784700c2e182f452f39cacd7e89876e9","observation_id":"6161839f-97d8-4820-b17f-79b608a8c284","resolution":{"observed_at":"2026-08-07T11:21:08.605193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.401648Z","title":"Emotional speech synthesis with rich and granularized control,","venue":null,"work_id":"c9013927-b674-4a5f-85b4-03252cfbf9ae","year":2020},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.728143Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:a53d11c4e47875b80b9ca81098e69e773589214f5166a17f8ca78739f59ac534","observation_id":"256c2a04-a96e-4bb4-ae67-ecbeac21ffe0","resolution":{"observed_at":"2026-08-07T11:21:14.405471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.389841Z","title":"Emospeech: guiding fastspeech2 towards emotional text to speech,","venue":null,"work_id":"37c8f573-c8ce-450b-8360-d3953ca4a8de","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.786675Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:3b4e5a4f82a21f1702b2f938c70cda67db9bf998865efe02cd6fe8cade5fc89b","observation_id":"f993f611-a2c7-49cf-ac88-d101dd4d335b","resolution":{"observed_at":"2026-08-07T11:21:14.393376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05447","last_updated":"2017-11-28T02:07:43Z","snapshot_observed_at":"2026-07-06T06:09:32.998168Z","submitted_at":"2017-11-15T08:27:35Z","title":"Emotional End-to-End Neural Speech Synthesizer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05447","snapshot_observed_at":"2026-08-07T11:21:08.876739Z","title":"Emotional end-to-end neural speech synthesizer,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.876739Z"},"links":{"cited_paper":"/paper/1711.05447","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:75eb1dac1e4b8848a18242c4e66abd70df67812a24ded10e78107b1fa51feb42","observation_id":"6fe71321-f5c4-45f9-9aa5-b5428c8438d4","resolution":{"observed_at":"2026-08-07T11:21:08.876739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18398","last_updated":"2025-02-18T21:39:25Z","snapshot_observed_at":"2026-07-06T18:06:49.190172Z","submitted_at":"2024-04-29T03:19:39Z","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18398","snapshot_observed_at":"2026-08-07T11:21:08.945087Z","title":"Mm-tts: A unified framework for multimodal, prompt-induced emotional text-to- speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.945087Z"},"links":{"cited_paper":"/paper/2404.18398","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:3d32715c638a14bad90fe7949ebefacff68cd8b973b73582005d290e2b2559ab","observation_id":"f07309b6-b2ef-48d8-b8e6-9e074049233b","resolution":{"observed_at":"2026-08-07T11:21:08.945087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.377719Z","title":"Emodiff: Intensity controllable emotional text-to-speech with soft-label guidance,","venue":null,"work_id":"19a6aa7f-b8d4-4bee-bc99-5e3cda16dd81","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.022798Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:1b655e66592a15a478cf5b0049b335f3ba44b4b9271567361a4f6571a263dffa","observation_id":"1a1a03c4-ca9c-4191-b63d-0ab860352e53","resolution":{"observed_at":"2026-08-07T11:21:14.381827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:09.117066Z","title":"Instructtts: Modelling expressive tts in discrete latent space with natural language style prompt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.117066Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b2567304ad8e0cc5c07a67dbd94f3ff29eacf23b46dcbd8add987e0e5b76459c","observation_id":"6cb213e9-216c-4e58-8c47-b84641a8b179","resolution":{"observed_at":"2026-08-07T11:21:09.117066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.358605Z","title":"Emotional voice conversion using dual supervised adversarial networks with continuous wavelet transform f0 features,","venue":null,"work_id":"fb3513d6-6d20-4d58-ae71-45c9094b52f4","year":2019},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.168520Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:ce0772797ecb68a438cfe6ac3ee85fed2c1d0f548abea48cd6bf7c885a050f5d","observation_id":"2893bc0f-cf81-40fb-a624-1adc0a30a85d","resolution":{"observed_at":"2026-08-07T11:21:14.361993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.347185Z","title":"Emotion controllable speech synthesis using emotion-unlabeled dataset with the assistance of cross-domain speech emotion recognition,","venue":null,"work_id":"4c68d45c-425d-4954-a061-9c904c5b7fab","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.260148Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:e2fd09cb550331a98e6ec63a74ea8f2f76fed60964d5a629f7355e9c2eb1425a","observation_id":"adf632af-02ca-46bf-9bcd-7d4ec723cb9b","resolution":{"observed_at":"2026-08-07T11:21:14.351042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12229","last_updated":"2024-09-17T10:40:11Z","snapshot_observed_at":"2026-07-06T18:47:33.863041Z","submitted_at":"2024-07-17T00:54:15Z","title":"Laugh Now Cry Later: Controlling Time-Varying Emotional States of Flow-Matching-Based Zero-Shot Text-to-Speech","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12229","snapshot_observed_at":"2026-08-07T11:21:09.341844Z","title":"Laugh now cry later: Controlling time-varying emotional states of flow-matching-based zero-shot text-to- speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.341844Z"},"links":{"cited_paper":"/paper/2407.12229","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:bdc5968b185e4603d11fca1afb476f96d72e1b394caffced0968f0855b36e8cd","observation_id":"60428489-0cc4-4ea2-a7a0-e3a977f91521","resolution":{"observed_at":"2026-08-07T11:21:09.341844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.00436","last_updated":"2022-10-06T07:47:16Z","snapshot_observed_at":"2026-08-07T18:37:13.571123Z","submitted_at":"2021-04-01T12:40:07Z","title":"Expressive Text-to-Speech using Style Tag","version":2},"cited_work":{"arxiv_id":"2104.00436","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.00436","snapshot_observed_at":"2026-08-07T11:21:12.472939Z","title":"Expressive Text-to-Speech using Style Tag","venue":"eess.AS","work_id":"60cef777-e48e-430c-bc98-a78d8e526aef","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.399942Z"},"links":{"cited_paper":"/paper/2104.00436","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:7c875d100a04fbf46a164b3ee39102d4e42fb4ced0538b83c1b846a2baa8a06a","observation_id":"3c9a0693-f18e-4c09-9302-fe359de62c9e","resolution":{"observed_at":"2026-08-07T11:21:12.618602Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.335541Z","title":"Controllable emotion transfer for end-to-end speech synthesis,","venue":null,"work_id":"261fba5f-8203-472b-afc1-fea830a66362","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.472161Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:a1d378e04cf8b01a9981a81c15e6bd5dd9d27bcf7d9ffc1ad3196c2474f9a07f","observation_id":"b68b8e23-6532-4930-b0e2-83071a623862","resolution":{"observed_at":"2026-08-07T11:21:14.339093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15185","last_updated":"2023-12-23T07:46:55Z","snapshot_observed_at":"2026-08-05T01:11:36.813583Z","submitted_at":"2023-12-23T07:46:55Z","title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15185","snapshot_observed_at":"2026-08-07T11:21:09.566555Z","title":"emotion2vec: Self-supervised pre-training for speech emotion repre- sentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.566555Z"},"links":{"cited_paper":"/paper/2312.15185","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:735e38775788aa697ba01b14e06e91d705b3e2d702ee94ddbb2dec8159d9bb6c","observation_id":"17b384e8-dc7f-4680-a281-53e03cb3b570","resolution":{"observed_at":"2026-08-07T11:21:09.566555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.323423Z","title":"Ttslow: Slow down text-to-speech with efficiency robustness evaluations,","venue":null,"work_id":"89d7f4f3-d2a2-4577-a8b9-4a7005fb3b64","year":2025},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.625831Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b44824abb326a253e6481f385be077200cd0e575490c89a6f1a55ebac3091a40","observation_id":"083825dc-034c-4b32-aad5-95ecadc7bf62","resolution":{"observed_at":"2026-08-07T11:21:14.327686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12498","last_updated":"2025-06-22T16:51:47Z","snapshot_observed_at":"2026-07-06T20:08:16.471575Z","submitted_at":"2024-12-17T03:02:05Z","title":"Hierarchical Control of Emotion Rendering in Speech Synthesis","version":3},"cited_work":{"arxiv_id":"2412.12498","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.12498","snapshot_observed_at":"2026-08-07T11:21:12.229573Z","title":"Hierarchical Control of Emotion Rendering in Speech Synthesis","venue":"cs.SD","work_id":"7e93314b-8382-4d9f-8a27-cb4427e71650","year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.698405Z"},"links":{"cited_paper":"/paper/2412.12498","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:0a44f7a5f3c342e559618c4451da3ff6d745d0915cfca21858321ca30bdc3f4d","observation_id":"4473eeaa-47f2-4a02-bada-4098ec1aec00","resolution":{"observed_at":"2026-08-07T11:21:12.319458Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.312310Z","title":"The nature of emotions: Human emotions have deep evolutionary roots, a fact that may explain their complexity and provide tools for clinical practice,","venue":null,"work_id":"4dde570c-c8ab-4451-9242-7cbeaedefd41","year":2001},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.755022Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:7a323fe10df7554b896e0f884459afca511760a7143bcd6773adc0926869d7f4","observation_id":"74199a7f-3414-4434-aa5e-12c14f82761b","resolution":{"observed_at":"2026-08-07T11:21:14.316055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.301327Z","title":"Mixed emotions and coping: The benefits of secondary emotions,","venue":null,"work_id":"b342b21f-8cc0-46ac-9fac-9028179aebec","year":2014},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.827686Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:0272e7aa67eab1f23990835b518e39579c4bac7500881798f54c29361a0497d9","observation_id":"705b1bd0-4113-4a5a-ab4f-9d2b36ff4cf6","resolution":{"observed_at":"2026-08-07T11:21:14.304965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00648","last_updated":"2023-06-01T13:14:56Z","snapshot_observed_at":"2026-07-06T15:36:33.788201Z","submitted_at":"2023-06-01T13:14:56Z","title":"EmoMix: Emotion Mixing via Diffusion Models for Emotional Speech Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00648","snapshot_observed_at":"2026-08-07T11:21:09.901971Z","title":"Emomix: Emotion mixing via diffusion models for emotional speech synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.901971Z"},"links":{"cited_paper":"/paper/2306.00648","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:6bbef7af75d9562334db736226dd8409caf879bdf4ca0ba698f1e4249f37b078","observation_id":"4bc94bfc-c0c3-4c33-b7a0-fcc02be94234","resolution":{"observed_at":"2026-08-07T11:21:09.901971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.285295Z","title":"Speech synthe- sis with mixed emotions,","venue":null,"work_id":"4f8f8dca-d3d4-4cda-9598-eff27847471c","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.007123Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:664bf1cf18f2cd085925c6426a94612d95948767f64d23330323dc16da411eb0","observation_id":"637c7ca9-a238-459a-b47b-d1b966943277","resolution":{"observed_at":"2026-08-07T11:21:14.293890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.980370Z","title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis,","venue":null,"work_id":"b0ec0d2b-ff29-4154-92a3-60de6447e699","year":2018},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.082748Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:263062ed318d53449c88619d23a1b60871d3fc42cc3880bdb05d33bacc272665","observation_id":"e6bc07f3-ae9e-48a9-90e5-0a18379650d9","resolution":{"observed_at":"2026-08-07T11:21:14.148867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.911370Z","title":"Fine-grained emotion strength transfer, control and prediction for emotional speech synthesis,","venue":null,"work_id":"7a851078-fd67-4378-ae7d-d2aba5b93249","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.142638Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:96c2d776027b97efc40e479d0afccac3230170d4b454d3ab61a0f69683c90be5","observation_id":"9c7fe274-f540-4bab-b00a-83df7318befd","resolution":{"observed_at":"2026-08-07T11:21:13.949555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.670011Z","title":"Emotion rendering for conversational speech synthesis with heterogeneous graph-based con- text modeling,","venue":null,"work_id":"4a8d7b74-446b-4d45-a3d0-4c7d002ff319","year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.225662Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:9e4436f7e74fa5a18067142ced1680f0c73f418afe7a72e00710e5fcc2e4b4e7","observation_id":"2497f4b3-b964-42be-9c88-309eb5c181b8","resolution":{"observed_at":"2026-08-07T11:21:13.784289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.409644Z","title":"An emotion speech synthesis method based on vits,","venue":null,"work_id":"5f9a2a9b-aa00-4f65-963b-225053f63c15","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.301961Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:d489d16f022a61515a32eefdd2221d5965df77788590e731764e617075a3326b","observation_id":"7d54fd94-5857-44f7-b432-56a2469d411b","resolution":{"observed_at":"2026-08-07T11:21:13.538624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:10.356461Z","title":"Emotional dimension control in language model-based text-to-speech: Spanning a broad spectrum of human emotions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.356461Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:1e05d09a58b7a92fce2543b21a3de27e7a3dd0a45c2058e5c9f466a4a88420e7","observation_id":"755c1a2e-765d-4e1d-9ca4-84c51f3e0796","resolution":{"observed_at":"2026-08-07T11:21:10.356461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10157","last_updated":"2024-09-16T10:41:36Z","snapshot_observed_at":"2026-07-06T19:15:56.947205Z","submitted_at":"2024-09-16T10:41:36Z","title":"Emo-DPO: Controllable Emotional Speech Synthesis through Direct Preference Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10157","snapshot_observed_at":"2026-08-07T11:21:10.425885Z","title":"Emo- dpo: Controllable emotional speech synthesis through direct preference optimization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.425885Z"},"links":{"cited_paper":"/paper/2409.10157","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:79ddd1b4933aa4667fe3fd440e6e53f49564aa146ad76c93a2fe66717041ec96","observation_id":"ea9f1639-a2a7-4b3d-9df5-8f402343c328","resolution":{"observed_at":"2026-08-07T11:21:10.425885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04128","last_updated":"2025-02-22T11:32:13Z","snapshot_observed_at":"2026-08-07T17:49:42.104594Z","submitted_at":"2025-02-06T15:04:00Z","title":"Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04128","snapshot_observed_at":"2026-08-07T11:21:10.561460Z","title":"Llasa: Scaling train-time and inference- time compute for llama-based speech synthesis,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.561460Z"},"links":{"cited_paper":"/paper/2502.04128","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:53e1c0381da710660cb651e98ce84974768e56515abc307c69b824fffe8ebf68","observation_id":"31081eed-7d6b-4f11-be56-b8fbb0f0f956","resolution":{"observed_at":"2026-08-07T11:21:10.561460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.197247Z","title":"Emo- dpo: Controllable emotional speech synthesis through direct preference optimization,","venue":null,"work_id":"5df9af8c-cd54-463e-b195-0e30e68bc4a4","year":2025},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.702413Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b5611983c56676e567e450e8968c1e4545eb216f160df1dd94307cd76151f98f","observation_id":"c0401735-66f1-420b-8b9c-bf5691d04bb4","resolution":{"observed_at":"2026-08-07T11:21:13.287159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:10.850442Z","title":"Plutchik and H","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.850442Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:7e7df833ca7341222f5299390aaccd7943b0e4de16c10e64c21a23d55522da37","observation_id":"81984663-2d80-4764-902d-f31aca722a97","resolution":{"observed_at":"2026-08-07T11:21:10.850442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:12.953074Z","title":"Cross and C","venue":null,"work_id":"2514adf7-b86d-4ff4-980d-97872902b527","year":2016},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.951853Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:453b3f5cbb0cde5a4fec4c378fdd2220a0d874231f02db7efa89e89f20f229b6","observation_id":"bae3163c-c55d-4294-958b-5ca37cfbe1ec","resolution":{"observed_at":"2026-08-07T11:21:13.069996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.13756","last_updated":"2023-09-18T01:54:31Z","snapshot_observed_at":"2026-08-06T02:42:18.339839Z","submitted_at":"2022-10-25T03:50:34Z","title":"Mixed-EVC: Mixed Emotion Synthesis and Control in Voice Conversion","version":3},"cited_work":{"arxiv_id":"2210.13756","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.13756","snapshot_observed_at":"2026-08-07T11:21:11.776569Z","title":"Mixed-EVC: Mixed Emotion Synthesis and Control in Voice Conversion","venue":"eess.AS","work_id":"bfaecd31-5ac8-4fa0-9ccb-f0d9a4b24a41","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.966398Z"},"links":{"cited_paper":"/paper/2210.13756","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:4d23a451325003e2744f66a66e6f1ccae9a613c1c9e9312afb52d3ced6b9b045","observation_id":"c5aa09e8-0832-4615-8684-2e077fc93150","resolution":{"observed_at":"2026-08-07T11:21:11.919290Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T11:21:11.058975Z","title":"GPT-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.058975Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:9b18a0ded03c722abd728664f83cd26be491f9ffd8adb167596c64819cae3578","observation_id":"c99d1b9d-5204-4302-a65a-863817ae65f3","resolution":{"observed_at":"2026-08-07T11:21:11.058975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T11:21:11.145489Z","title":"Gemini: a family of highly capable multimodal models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.145489Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:d9645e6ff1b4bd33fb95bd96aef4bcbb103c285c8d2c9733a74064b1059e9cd5","observation_id":"3ac0bed4-f7b3-4dea-96ab-d39ab32481a3","resolution":{"observed_at":"2026-08-07T11:21:11.145489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T11:21:11.313415Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.313415Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:5bb34a044486edd651a9973d18381d66fb3cd448603e64fa915bfadf417f00a3","observation_id":"3943616c-b04d-4732-b3eb-f60faec16bfc","resolution":{"observed_at":"2026-08-07T11:21:11.313415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:12.775881Z","title":"Emotional voice conversion: Theory, databases and esd,","venue":null,"work_id":"f5ac3f13-2bd6-403d-82a5-12cce5ec57dd","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.425075Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:71ab3728dbeffe577eee6c10736e1a9db1efc3cb3a53acbc460bf891566fcc5e","observation_id":"b63597ec-abaa-4a80-9499-ab4411a7ab91","resolution":{"observed_at":"2026-08-07T11:21:12.866746Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-07T11:21:11.541084Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.541084Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:cdc515f54109c3a34254b153533d75a5daf9055b688e8c0ef113d08ab57b6ae7","observation_id":"000021e7-1a98-488e-a767-fbe79ea35c0a","resolution":{"observed_at":"2026-08-07T11:21:11.541084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-07T15:00:33.370493Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":3,"verified_fuzzy":21},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 1 inbound Pith citation observation for arXiv:2506.02742."}