{"as_of":"2026-08-12T16:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a20de5c61694fed00ef86a35c272792e9710799beae8bb01f1cf509753645b2e","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:29:43.405193Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.04904/citation-record","integrity":"/paper/2501.04904/integrity","json":"/paper/2501.04904/citation-record.json","paper":"/paper/2501.04904"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.874239Z","title":"Natural TTS Synthesis by Conditioning Wavenet on Mel Spectrogram Predictions,","venue":null,"work_id":"98475047-87cc-4d15-8a3a-3a4361b52af9","year":2018},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.250452Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:5101898df18a7042071d2cd65648e0fffaa6050ac63f78bddb0498a989899e99","observation_id":"455ed74b-dcef-479c-9e4e-35ffe4ac91b9","resolution":{"observed_at":"2026-08-10T21:29:43.878455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.862268Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech,","venue":null,"work_id":"edd3d284-e07a-4534-9c62-153a67a060c7","year":2021},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.254928Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:5b75adf67ac698fb62532201bcc12eb02fe0052bc3a8cb9fa4149a8232e7e53a","observation_id":"5775c4de-2f6f-4eb9-989d-47f62652191c","resolution":{"observed_at":"2026-08-10T21:29:43.866233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.849404Z","title":"HierSpeech: Bridging the Gap between Text and Speech by Hierarchical Variational Inference using Self-supervised Representations for Speech Synthesis,","venue":null,"work_id":"38ce3464-99fe-4a8e-b7b6-c19c06e20ae5","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.258540Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:3dcd59b815516ec092eca12bb586f544d1c0949bab7b0dd2293294db7bdad7f1","observation_id":"e7e85684-bfa0-4440-a5d5-4eb2e7cadf91","resolution":{"observed_at":"2026-08-10T21:29:43.853674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.837122Z","title":"Matcha-TTS: A fast TTS Architecture with Conditional Flow Matching,","venue":null,"work_id":"f24a0155-8418-46eb-a834-936a0db3da52","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.262268Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:d3c0932561f3049d8023a6cdd604a935e3d6e09f23eaf87d12884ebf346cfba8","observation_id":"90b748e9-fa3a-4628-835d-8b599e50947f","resolution":{"observed_at":"2026-08-10T21:29:43.841512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.825490Z","title":"V oice- Flow: Efficient Text-To-Speech with Rectified Flow Matching,","venue":null,"work_id":"41a03475-dd07-45ff-b195-de4531894259","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.266357Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:af2bbcaaf93748cf2be72474734eb0e1a7e0afe7a30f54931cb32fa892b91203","observation_id":"e20bbb16-47d6-40d3-ba20-4e328af25522","resolution":{"observed_at":"2026-08-10T21:29:43.829346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.812850Z","title":"Conversational End-to-End TTS for V oice Agents,","venue":null,"work_id":"1fbf7a84-b7de-494e-ae7e-bed59e0773de","year":2021},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.269897Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:10de73c0f63a2ef1ceeef3e38f53ca56f83bf43beabeecb038e8693eebb55741","observation_id":"32510b16-9547-40a3-8b5f-d1a683c714f9","resolution":{"observed_at":"2026-08-10T21:29:43.817053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.799983Z","title":"Enhancing Speaking Styles in Conversational Text-to- Speech Synthesis with Graph-Based Multi-Modal Context Modeling,","venue":null,"work_id":"e62ee6c2-40d0-4e97-999d-555e9ad60f3a","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.273627Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:6d7f06a629a23d743920f9cd0a389bbccdd78d93260cfd313fe8520910f5e418","observation_id":"ba724b39-767d-444f-9a78-f4f13c82626b","resolution":{"observed_at":"2026-08-10T21:29:43.804447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.786929Z","title":"M2-CTTS: End-to-End Multi-Scale Multi-Modal Conversational Text-to-Speech Synthesis,","venue":null,"work_id":"57290c54-3797-4b78-a391-e779c77e4c41","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.276720Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:0e0a79124e97a5cb2284f2fd2fffa1abec2ecc757c4125cfee1400b4aaf76f4a","observation_id":"ee3e98f9-7070-44e0-931e-9431589ab592","resolution":{"observed_at":"2026-08-10T21:29:43.791811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.777108Z","title":"Concss: Contrastive-based Context Comprehension for Dialogue-Appropriate Prosody in Conversational Speech Synthesis,","venue":null,"work_id":"a9a3c139-72a3-42d1-9a0a-a3da71d4b7d4","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.280053Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:3d8df32a12f0911fcfb887529d62ca79965238ab29dc180c74b4c02d0efb0082","observation_id":"742097bf-bc69-4f58-8f98-1d2a851855ec","resolution":{"observed_at":"2026-08-10T21:29:43.780401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.768554Z","title":"Considering Temporal Connection between Turns for Conversational Speech Synthesis,","venue":null,"work_id":"6be81d14-5dc2-4b13-b011-f5871136a5c1","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.283619Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:fc1afc99c9b57376b2e66b91b967034c6ee9a0d7e02120a6c839a637b4e37010","observation_id":"5ce4b5e3-cffa-4fc9-9454-d64d312ae447","resolution":{"observed_at":"2026-08-10T21:29:43.771439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.758495Z","title":"A new recurrent neural-network architecture for visual pattern recognition,","venue":null,"work_id":"1ad02544-f9fe-4ef8-8450-5be2c760bdf3","year":1997},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.287141Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:b1855493b3cb9561b9c94a71b49f0e38dc88eefa0b80862c8398ad7bbc8b1491","observation_id":"fb2c1b8d-15ba-40ce-bab2-c8de5594d452","resolution":{"observed_at":"2026-08-10T21:29:43.762313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.747807Z","title":"Classification of drowsiness levels based on a deep spatio-temporal convolutional bidirectional LSTM network using electroencephalogra- phy signals,","venue":null,"work_id":"b779ae72-eac0-47b4-ae37-289211a719d4","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.290406Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:378d13606e8afdd871563d3d42b524ad03cf07767e0c4cb3d11f61eb60729d8f","observation_id":"3a40749b-a6f4-4121-b242-d43c29bd8288","resolution":{"observed_at":"2026-08-10T21:29:43.751693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.737099Z","title":"Towards an EEG-based intuitive BCI communication system using imagined speech and visual imagery,","venue":null,"work_id":"d5e85884-baae-4faa-a197-e6db406063ab","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.293873Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:c85cdd25ab3f2054b14a8de3f35c0bf40acc9ab2cc167aabd194479f3e9009da","observation_id":"beffdf87-2a96-4cb6-8721-f96253610690","resolution":{"observed_at":"2026-08-10T21:29:43.741084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.724560Z","title":"A multi-view cnn with novel variance layer for motor imagery brain computer interface,","venue":null,"work_id":"4a4b91ad-79ad-43d0-91f9-3ca11810bc25","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.297415Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:3398a7523a02986a249c056491f26ee30d5592cf8317441d1f4756769f773406","observation_id":"164e2906-c334-4023-94c1-e87a15e31658","resolution":{"observed_at":"2026-08-10T21:29:43.729368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.713972Z","title":"An adaptive deep reinforcement learning framework enables curling robots with human-like performance in real-world conditions,","venue":null,"work_id":"67f7eb1d-e164-4c72-a242-76b1e4f041fc","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.300931Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:9a3e4853fd8c4a4105c728d47bf0fc28492b29fa54c0f88e7f6b803e535e86ff","observation_id":"fcafa22c-1ec9-4a12-a8e9-a0e61124b666","resolution":{"observed_at":"2026-08-10T21:29:43.717871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07547","last_updated":"2024-08-14T13:36:17Z","snapshot_observed_at":"2026-07-06T19:00:40.259490Z","submitted_at":"2024-08-14T13:36:17Z","title":"PeriodWave: Multi-Period Flow Matching for High-Fidelity Waveform Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07547","snapshot_observed_at":"2026-08-10T21:29:43.304427Z","title":"Periodwave: Multi-period flow matching for high-fidelity waveform generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.304427Z"},"links":{"cited_paper":"/paper/2408.07547","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a5dad36df13bd7fdc1c65684c10656f1ada599bedf0cf7d5cc79295457ecb4d2","observation_id":"9c98865d-edeb-494d-960a-ebc03960a0c9","resolution":{"observed_at":"2026-08-10T21:29:43.304427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.703606Z","title":"Emoq-tts: Emotion intensity quantization for fine-grained controllable emotional text-to-speech,","venue":null,"work_id":"c324eefe-a835-4ede-bced-509408cb3f9b","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.308697Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:274f92e3f838196036c6c119284690e24bce36c9e848e0b6b10aad8d298d63cf","observation_id":"6ab269ed-6f19-44d1-88c0-899b507cf3f8","resolution":{"observed_at":"2026-08-10T21:29:43.707472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.692794Z","title":"Diffprosody: Diffusion-based latent prosody generation for expressive speech synthe- sis with prosody conditional adversarial training,","venue":null,"work_id":"41783aa6-25fd-4047-b455-23af23f4f08f","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.312310Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:cf381f137a24a5e77cc074c533dedc4a1fd6eba0af65d841eafd0564507184ef","observation_id":"6996cda2-e285-40fb-830d-fb887bd96225","resolution":{"observed_at":"2026-08-10T21:29:43.696860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.681980Z","title":"EmoSphere-TTS: Emotional Style and Intensity Modeling via Spherical Emotion Vector for Controllable Emotional Text-to-Speech,","venue":null,"work_id":"c82858ff-be03-4f47-9c92-c7eaecaac4e1","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.316220Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:769122d58e128f1b231d966f5a9b5d40cbaa38db86ebfc94b385e9f2e6c9abcd","observation_id":"984f69b8-44a3-4d20-8b4a-8b5c772f9de0","resolution":{"observed_at":"2026-08-10T21:29:43.686283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.08095","last_updated":"2025-01-21T02:51:53Z","snapshot_observed_at":"2026-08-12T11:08:00.154685Z","submitted_at":"2024-01-16T03:39:35Z","title":"DurFlex-EVC: Duration-Flexible Emotional Voice Conversion Leveraging Discrete Representations without Text Alignment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.08095","snapshot_observed_at":"2026-08-10T21:29:43.320033Z","title":"DurFlex-EVC: Duration-Flexible Emotional V oice Conversion with Parallel Generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.320033Z"},"links":{"cited_paper":"/paper/2401.08095","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:397c777a9c63146cefdf2924a675b6dd2ffab5c698d103726786f6e74a98526d","observation_id":"341a8fea-0dac-4930-8223-911bc5be0baf","resolution":{"observed_at":"2026-08-10T21:29:43.320033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.671968Z","title":"Emotion rendering for conversational speech synthesis with heterogeneous graph- based context modeling,","venue":null,"work_id":"28cd1abc-d0ab-4bae-9403-e547608cb909","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.324595Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:0aa2a1069be868d5dea62a6f797368ceb2995044d3aa8c7a8aa64c926796bd5a","observation_id":"94e8086e-f874-41c0-a17f-f6512c75925b","resolution":{"observed_at":"2026-08-10T21:29:43.675754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16420","last_updated":"2024-01-29T18:59:02Z","snapshot_observed_at":"2026-08-05T03:42:54.599829Z","submitted_at":"2024-01-29T18:59:02Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16420","snapshot_observed_at":"2026-08-10T21:29:43.328351Z","title":"Internlm-xcomposer2: Mastering free-form text- image composition and comprehension in vision-language large model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.328351Z"},"links":{"cited_paper":"/paper/2401.16420","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:df55b07c5f0bea3c857f11bb8e960a14e774120835fcc5cb672c77c732386b80","observation_id":"7e155f12-d29e-4db2-ac40-b730b9212873","resolution":{"observed_at":"2026-08-10T21:29:43.328351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19041","last_updated":"2024-05-29T12:32:08Z","snapshot_observed_at":"2026-08-03T00:02:53.450039Z","submitted_at":"2024-05-29T12:32:08Z","title":"BLSP-KD: Bootstrapping Language-Speech Pre-training via Knowledge Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19041","snapshot_observed_at":"2026-08-10T21:29:43.332430Z","title":"BLSP-KD: Bootstrapping Language-Speech Pre-training via Knowl- edge Distillation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.332430Z"},"links":{"cited_paper":"/paper/2405.19041","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:49dac1c85e097b6e61118045f6c125ac9b12d3d0fd112687dea3af1db8a1ac09","observation_id":"d4b4056f-d47d-4147-a27a-19bb8f70547a","resolution":{"observed_at":"2026-08-10T21:29:43.332430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.661126Z","title":"Whisper-AT: Noise-Robust Automatic Speech Recognizers are Also Strong General Audio Event Taggers,","venue":null,"work_id":"95eb563a-d53a-4b29-b987-a86b24a20274","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.336305Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:3ddabda52539f5e3a0754de9dcf9b5ddb33ae4b52d1be9af9b48e2672dfe3612","observation_id":"6e31169c-973e-422b-8cda-896d7e88e3e7","resolution":{"observed_at":"2026-08-10T21:29:43.665088Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.651661Z","title":"BLIP-2: Boot- strapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models,","venue":null,"work_id":"72db5829-77da-41e5-bf77-d6e4dc853c05","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.340004Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:8b1093bf28e0cd0b499668d5684033ea431d00b6231f1b5ac74caa0a48a488be","observation_id":"72786add-657c-47f6-a067-d0c711000053","resolution":{"observed_at":"2026-08-10T21:29:43.654723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.643098Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision,","venue":null,"work_id":"b2ff37b1-3066-4652-b7a0-73f7ca42d195","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.343722Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:33aa5046cd2ad8b77a21548913e62f8f7217965919711be926b6309d0de2d9c2","observation_id":"d6b51203-c677-478f-bc65-b3d7ff262897","resolution":{"observed_at":"2026-08-10T21:29:43.646248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.634437Z","title":"Joint Audio and Speech Understanding,","venue":null,"work_id":"bcb082b4-66dc-4966-ba69-a68faaeba9d8","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.347055Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:bd4227b91199994689d97b933e0a21043f9802cbf1b95dc85ab5b74849ae566b","observation_id":"2b3da1c3-3ed5-40f3-b68a-f60507af2742","resolution":{"observed_at":"2026-08-10T21:29:43.637554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.623884Z","title":"HiFi-GAN: Gen- erative Adversarial Networks for Efficient and High Fidelity Speech Synthesis,","venue":null,"work_id":"a1427cc5-5d26-4413-8ed2-554c4f14b038","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.350771Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:714db11f722d50789a06fcd4da37ce94ca8aaf159ea412fbcc98c05ef290b06b","observation_id":"4677b17d-40d3-4301-8b47-3c7cf683dc18","resolution":{"observed_at":"2026-08-10T21:29:43.627949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.613427Z","title":"DailyTalk: Spoken Dialogue Dataset for Conversational Text-to-Speech,","venue":null,"work_id":"ccc6e73a-67b8-4bec-adff-8ac96edd5c4c","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.354260Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:f58f1df0303a0bad7863b25f86bb38128fc8dcc060e193318cf3f53ea10735a2","observation_id":"147b7bcf-18f5-4439-b418-170a3605770e","resolution":{"observed_at":"2026-08-10T21:29:43.617661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.603042Z","title":"DailyDialog: A Manually Labelled Multi-turn Dialogue Dataset,","venue":null,"work_id":"5cc177db-7b27-44b8-88b9-4a5ccab1fa62","year":2017},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.357664Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:fb07801d2282eec49981c2478497fe1ed2040a77d3886318f04a6c4ac4e9ef3e","observation_id":"f0ae0780-a0db-4eb4-a9f3-39cd9a8dff17","resolution":{"observed_at":"2026-08-10T21:29:43.607229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.592353Z","title":"CREMA-D: Crowd-Sourced Emotional Multimodal Actors Dataset,","venue":null,"work_id":"189872cb-0f17-4740-96e1-862e144955b9","year":2014},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.361632Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:03a10b324e5354459e6f1af9755a6d6a20694ea8e76b95d1bb86e5751fe8a126","observation_id":"52c67a9b-2bec-4a87-9803-ea1867d58737","resolution":{"observed_at":"2026-08-10T21:29:43.596339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1806.09514","last_updated":"2018-06-25T15:01:54Z","snapshot_observed_at":"2026-08-10T08:17:27.164286Z","submitted_at":"2018-06-25T15:01:54Z","title":"The Emotional Voices Database: Towards Controlling the Emotion Dimension in Voice Generation Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.09514","snapshot_observed_at":"2026-08-10T21:29:43.365529Z","title":"The emotional voices database: Towards controlling the emotion dimension in voice generation systems,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.365529Z"},"links":{"cited_paper":"/paper/1806.09514","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:5ad407d814eb01358b06bd7282575a79b061155a0be8e814f9b599e3a8bad676","observation_id":"826908ef-0d70-4fe4-9dcb-0992498ea9d1","resolution":{"observed_at":"2026-08-10T21:29:43.365529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.580829Z","title":"IEMOCAP: Interactive emotional dyadic motion capture database,","venue":null,"work_id":"b207d8b2-f048-48a2-9f0e-47293c6e052d","year":2008},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.369873Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:ac5e3dd778d26d58d38084dd43917c2316289f943b561e3ae236be3c667009ca","observation_id":"641c9827-6c87-4460-9c1b-7c46aef6d3cc","resolution":{"observed_at":"2026-08-10T21:29:43.585460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.571082Z","title":"MEAD: A Large-scale Audio-visual Dataset for Emotional Talking-face Generation,","venue":null,"work_id":"a80fc278-e88e-455b-93eb-aab28cefa249","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.373630Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:701e6d0f4254caa56fc09b5d8cb6f155d12cfa50fa936b4e3eba943b1327d932","observation_id":"7171c184-92f1-43ca-8ef0-bf1fc0f118e5","resolution":{"observed_at":"2026-08-10T21:29:43.574845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.560055Z","title":"Toronto emotional speech set (tess)-younger talker happy,","venue":null,"work_id":"113f0e51-ff71-4b46-8159-c8fbcbcf30c3","year":2010},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.377953Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:3b6c720c181f0d951fe6e28136e1b485bdbe7c95f4294f6c12d554cf6ae42199","observation_id":"6c25328d-a94d-4d36-a1fe-da76096e92e1","resolution":{"observed_at":"2026-08-10T21:29:43.564236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.548897Z","title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality,","venue":null,"work_id":"04d1dcf3-8de1-4f46-895e-7bd7125d40b4","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.381638Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:f7bd32b74e8b3e052ee094d4d9a959b9f8e1266862b6a9277e53e1823fff25d3","observation_id":"2c0d3525-bab5-4ed6-b275-f2e7825b8c83","resolution":{"observed_at":"2026-08-10T21:29:43.552689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-10T21:29:43.385693Z","title":"LLama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.385693Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:41fdd9f7dc4237c89623e13327bdd325bb825267e8998fd7f01fd174e428c0c5","observation_id":"4def9cc2-dc2c-448f-ae21-4702a0626c30","resolution":{"observed_at":"2026-08-10T21:29:43.385693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.538699Z","title":"Decoupled Weight Decay Regular- ization,","venue":null,"work_id":"4de78178-11d7-4b9b-b382-52e5eb80bfc3","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.389703Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:6bab99b5f173eb01e1e17587bb1a32318f560fece93210dcff9fddea7bbf2a87","observation_id":"7648c2f8-f3da-4d9d-ae90-fab86882c09a","resolution":{"observed_at":"2026-08-10T21:29:43.542313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.528162Z","title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation,","venue":null,"work_id":"3f050773-dced-4b67-b571-dbcf46034b8a","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.393629Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:2c574e019999e40c8e4101ef44a1b4eb55c6aaee84f4ab0b55e552331f4307f4","observation_id":"67b43322-bc93-41fb-b1cf-9886ae39087c","resolution":{"observed_at":"2026-08-10T21:29:43.531861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.517821Z","title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations,","venue":null,"work_id":"222d9879-1fae-49e6-9ec2-04e6d440e976","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.397315Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:549b5243bb57671ce8050a528e457a86bc3b91652416559d001313e6c271d426","observation_id":"5f18c2de-8774-4623-9571-2f397f39f659","resolution":{"observed_at":"2026-08-10T21:29:43.521601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.507564Z","title":"Sequence-to-Sequence Acoustic Modeling for V oice Conversion,","venue":null,"work_id":"0acf2324-0bf5-4c79-8a6c-55c947015f2c","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.401154Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a5aab246a8e1e804259237242f03eaebed67fbe8b69b00c86562a14a6cd6ddea","observation_id":"9c39e78d-c66c-41b2-9c47-f32b48e68740","resolution":{"observed_at":"2026-08-10T21:29:43.511085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.494193Z","title":"LoRA: Low-Rank Adaptation of Large Language Models,","venue":null,"work_id":"65556e09-ef71-407a-993f-8fbae8fb6eba","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.405193Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:ad33cc9798c30d4335b90caa6b46a9fecc716ebca358f2ddc882c12daefa4dc6","observation_id":"4df5316b-c29f-4c04-9415-635dde855aaa","resolution":{"observed_at":"2026-08-10T21:29:43.499432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-12T11:08:56.513596Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":6,"verified_exact":0,"verified_fuzzy":36},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2501.04904."}