{"as_of":"2026-08-09T22:40:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:be3b7bc64d7d5d748d63bd3078e186a5e5ea89e0061a808359ff663519920a64","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T01:58:26.876241Z","state":"measured"},{"denominator":65,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":65,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.15755/citation-record","integrity":"/paper/2607.15755/integrity","json":"/paper/2607.15755/citation-record.json","paper":"/paper/2607.15755"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.07676","last_updated":"2024-06-11T19:50:50Z","snapshot_observed_at":"2026-07-06T18:29:12.931702Z","submitted_at":"2024-06-11T19:50:50Z","title":"FastAST: Accelerating Audio Spectrogram Transformer via Token Merging and Cross-Model Knowledge Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07676","snapshot_observed_at":"2026-08-03T01:58:20.944864Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:20.944864Z"},"links":{"cited_paper":"/paper/2406.07676","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:f8bbace98607ff1d3bc681d1dd97b58ebb86927457fb9513035beb5bde513321","observation_id":"ee07aa5e-bf4b-4b06-a50a-1dce7974fe14","resolution":{"observed_at":"2026-08-03T01:58:20.944864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09461","last_updated":"2023-03-01T19:45:11Z","snapshot_observed_at":"2026-07-06T14:06:56.291161Z","submitted_at":"2022-10-17T22:23:40Z","title":"Token Merging: Your ViT But Faster","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.09461","snapshot_observed_at":"2026-08-03T01:58:20.994874Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:20.994874Z"},"links":{"cited_paper":"/paper/2210.09461","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:8a6d68b1c914dc7912d83ee583ed97c14447c9469a846ee15581c233ad260667","observation_id":"20570d17-79af-453b-8c81-a96d8abad5e7","resolution":{"observed_at":"2026-08-03T01:58:20.994874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:21.075073Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.075073Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:e7c63526897ede7ad0d5c88cfebabb2c1ad45e8e4368ecd397a6cca3da471737","observation_id":"353298be-5664-4552-b046-88b1a984487b","resolution":{"observed_at":"2026-08-03T01:58:21.075073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-03T01:58:21.114068Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.114068Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:a1aecb0f6c1394e4910a0c5ee64e118e6634563e432003076f1b6fdd36833bb0","observation_id":"c82404ff-e4d2-4aec-bd2a-9dd1ab7991ee","resolution":{"observed_at":"2026-08-03T01:58:21.114068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.3555","last_updated":"2014-12-11T06:46:53Z","snapshot_observed_at":"2026-08-03T11:40:51.182181Z","submitted_at":"2014-12-11T06:46:53Z","title":"Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.3555","snapshot_observed_at":"2026-08-03T01:58:21.235022Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.235022Z"},"links":{"cited_paper":"/paper/1412.3555","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:9c196f652bf6c249246b9a53f5aefd2e041eaa34083f319934009e2069b2a55c","observation_id":"367ac547-0b9f-4944-a776-35180792ea2c","resolution":{"observed_at":"2026-08-03T01:58:21.235022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-03T01:58:21.431206Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.431206Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:62d1eb1ce9cc72f6bca66020419efc10efd473188e202927b2c004153f31f621","observation_id":"1d49be55-5acf-4164-bb52-dd0ab8ffa0d9","resolution":{"observed_at":"2026-08-03T01:58:21.431206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:21.536933Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.536933Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:2bbdc5f31a250da233c8c1e6be3f16722260c0631f171316c2fcd79dd965ad9b","observation_id":"1c733b6f-a202-4748-8664-119eedbf8860","resolution":{"observed_at":"2026-08-03T01:58:21.536933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:21.768393Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.768393Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:081ac56aca9be576cad8e876680cbbff3b06b7604454de63d54e880094a1df90","observation_id":"a65c640b-b506-4cef-92de-cd1b850a96a4","resolution":{"observed_at":"2026-08-03T01:58:21.768393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.18425","snapshot_observed_at":"2026-08-03T01:58:21.974177Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:21.974177Z"},"links":{"cited_paper":"/paper/2504.18425","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:92304f6c15f7795bf251ac69d52a666011cf935bd3bf9782267865eaf93d2985","observation_id":"79179b28-cdf0-4c74-a77b-a96314849cdf","resolution":{"observed_at":"2026-08-03T01:58:21.974177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-03T01:58:22.186072Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.186072Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:399ddaf29477da6aef2828d170d2389ddb72937e12c2b222afaf85ad1f00a736","observation_id":"ed392194-110a-4f00-97e9-0e3659bbe923","resolution":{"observed_at":"2026-08-03T01:58:22.186072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-09T04:36:59.879757Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-03T01:58:22.289313Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.289313Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:4e4b536221f0931d99264605d57a453e45078f854b1ea72f55ca507297512eec","observation_id":"3bc3b2d2-8f3c-413c-be24-d5c497d04b5e","resolution":{"observed_at":"2026-08-03T01:58:22.289313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.356795Z","title":null,"venue":null,"work_id":null,"year":1992},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.356795Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:f5d7f42968a1136ddfc705d9792d9831d9e8ec7fa07fa7060af685aaa63b45bd","observation_id":"24c302d3-4671-45d5-9a23-cf7858cf10ba","resolution":{"observed_at":"2026-08-03T01:58:22.356795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.433663Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.433663Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:7940e09dce3c476342f151086a9fbece997558424da13002f0cb3d5803b4c93d","observation_id":"09397ebd-18ab-4ca3-9b7b-e0edc8ab94a5","resolution":{"observed_at":"2026-08-03T01:58:22.433663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.16144","last_updated":"2026-04-20T04:32:07Z","snapshot_observed_at":"2026-07-06T22:46:12.384926Z","submitted_at":"2026-02-18T02:29:33Z","title":"Missing-by-Design: Certifiable Modality Deletion for Revocable Multimodal Sentiment Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.16144","snapshot_observed_at":"2026-08-03T01:58:22.496251Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.496251Z"},"links":{"cited_paper":"/paper/2602.16144","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:46444474619fd0d9e7c3e9f8f4c09d942dded572b6c6a003023d37a6b8eecbb4","observation_id":"882ef7ac-6483-4426-b02b-5114e1829164","resolution":{"observed_at":"2026-08-03T01:58:22.496251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.556914Z","title":null,"venue":null,"work_id":null,"year":1994},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.556914Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:a735e3f1d68db1a5c67d5f52a3449439ef30df237afecb0e3df4e666250519a2","observation_id":"4c349a98-1a28-4dcd-a506-492c2d792394","resolution":{"observed_at":"2026-08-03T01:58:22.556914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.620587Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.620587Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:d5927c7d63ed7b0ef84555c523e82ffcafc102274f322863122956e30de7ac5e","observation_id":"e03562f4-1995-4fd1-aa75-3510e965f0f5","resolution":{"observed_at":"2026-08-03T01:58:22.620587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.681504Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.681504Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:44c67f008a17b547089686c2acbd5c77f028c825b05edc29286fb11380cc5dff","observation_id":"85cbde60-1298-4b72-941b-30146f83c978","resolution":{"observed_at":"2026-08-03T01:58:22.681504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.738957Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.738957Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:0e8ab3ac7deded349c69f77b6ad9ad3c62c98c4d870f41870bce09a7e5ba4284","observation_id":"81e5f2f1-ac8c-4bc2-9531-d07a4f4a43ee","resolution":{"observed_at":"2026-08-03T01:58:22.738957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.844276Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.844276Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:1a5d83561ca6b40ff19831a052029e134d5157a5f6018699006d200723a7b1e3","observation_id":"064f3660-3a1d-4d84-9561-4264d603df0a","resolution":{"observed_at":"2026-08-03T01:58:22.844276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:22.952371Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:22.952371Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:28ded226a60aab263b21a60e463cb578359849c092bd1360968f08387acc5692","observation_id":"b7c207ff-c7d7-4884-a60e-70468d7fbee0","resolution":{"observed_at":"2026-08-03T01:58:22.952371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.028835Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.028835Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:a2dc975cef25288782ed36c1537103238166064012619bf1b31257bddf2a1bc4","observation_id":"48d5e23e-cc7e-4298-9aeb-30f7bd14fef5","resolution":{"observed_at":"2026-08-03T01:58:23.028835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03100","last_updated":"2024-04-23T08:38:03Z","snapshot_observed_at":"2026-08-07T23:22:31.422843Z","submitted_at":"2024-03-05T16:35:25Z","title":"NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03100","snapshot_observed_at":"2026-08-03T01:58:23.094743Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.094743Z"},"links":{"cited_paper":"/paper/2403.03100","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:db6c9122bb3e349868a3195c7d81c0e39be1d7ae518fc7dd42127ed467cdaa9e","observation_id":"00f8d0ee-79ca-4365-8393-92cebc2d0277","resolution":{"observed_at":"2026-08-03T01:58:23.094743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.244744Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.244744Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:3a89bfdd7b87859d46f8a8bca679733a429ae73f2a83790cd3f93a98c022c47a","observation_id":"4b2bf7cb-47e5-48dc-b05f-7951527c2861","resolution":{"observed_at":"2026-08-03T01:58:23.244744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.434839Z","title":null,"venue":null,"work_id":null,"year":1993},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.434839Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:8c30a0195b8ea5296fecbff223e83031b4bf7c0a9a89fd989a28191bdbba559b","observation_id":"f58e6449-9ce2-472b-8de3-746fae036c1d","resolution":{"observed_at":"2026-08-03T01:58:23.434839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.544747Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.544747Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:f21da75ab5c084786acd7a50f31bc1edde2585fc1eea4f69915b7095b408ce65","observation_id":"4452f385-c357-4124-87e0-28e851448f52","resolution":{"observed_at":"2026-08-03T01:58:23.544747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.652256Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.652256Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:38c50641f935136d0ce23be79ca0536cf1a94ee26d9212bd409874211dab6e37","observation_id":"3c527123-dfc9-453c-bf01-826573e2d5ee","resolution":{"observed_at":"2026-08-03T01:58:23.652256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.762892Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.762892Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:65965f8d006a492a7f5ae4b5a29370bacd879ae8d6dcb3ef02a3a4d07493fd20","observation_id":"0e997580-f16d-468f-9fff-c3a341754356","resolution":{"observed_at":"2026-08-03T01:58:23.762892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:23.894751Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:23.894751Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:e98ebdd1c146ba39e3da17498a3881e306f7e9c34f41d94adfb652a761d28625","observation_id":"62290024-0232-4824-9d3f-699832462836","resolution":{"observed_at":"2026-08-03T01:58:23.894751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.017414Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.017414Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:0cbdc67852642244ca3842a7778328c1507f3610514c81d302d9e2b4de362f70","observation_id":"b27de8d1-d67a-495c-9927-eff0d860354c","resolution":{"observed_at":"2026-08-03T01:58:24.017414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.163042Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.163042Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:edc1b9f53dadafed560db10f8613b20688e327f61fce82402033806c630636ab","observation_id":"ecfbb710-89fb-4074-a661-7768d2398628","resolution":{"observed_at":"2026-08-03T01:58:24.163042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.319819Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.319819Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:1c3358d54fbbba666a997a6a88d225c1bcb7bce95d0777d5ce1c43c16773bbba","observation_id":"516ab9fd-971f-4883-b324-063006ef4584","resolution":{"observed_at":"2026-08-03T01:58:24.319819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.457093Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.457093Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:d0dafaeca3387363ca645c7310db60d6803d8b66c00308b453f18c2ed2c84684","observation_id":"c7af3305-e125-486c-9baa-45f57b3c3ebd","resolution":{"observed_at":"2026-08-03T01:58:24.457093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.516618Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.516618Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:3fde3d783ee4da9883b1833a8085f054127d31143b2532161d5bf4a4b142589a","observation_id":"6f784bf8-07cc-414d-ab2a-b88a6b4bf4c0","resolution":{"observed_at":"2026-08-03T01:58:24.516618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09524","last_updated":"2024-10-12T13:02:31Z","snapshot_observed_at":"2026-08-08T14:33:06.261733Z","submitted_at":"2024-10-12T13:02:31Z","title":"Emphasis Rendering for Conversational Text-to-Speech with Multi-modal Multi-scale Context Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.09524","snapshot_observed_at":"2026-08-03T01:58:24.594826Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.594826Z"},"links":{"cited_paper":"/paper/2410.09524","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:c48d647d275c35e7c85984a25de0547dfed9b509ad5210717123f72c305ef8c2","observation_id":"f4ce2ccb-f563-4e8b-a927-64e26552e71a","resolution":{"observed_at":"2026-08-03T01:58:24.594826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.632113Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.632113Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:28853f5e505744a15c82c3e978c086ad18a5f52c9bd8d1aee14c1ebbb6ca8f49","observation_id":"739885b6-b292-481e-8df2-3fbf4c5315d1","resolution":{"observed_at":"2026-08-03T01:58:24.632113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.711982Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.711982Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:8bfc1306f724b161c27750e861030f5310db27651cc1c6e891cee62ad6f171d8","observation_id":"3fcd7099-2332-44cb-b235-60dee42be456","resolution":{"observed_at":"2026-08-03T01:58:24.711982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.778065Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.778065Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:540d7665894a59d997332a4621bca558cd0c6f9c858c1d7367a285a57b7cce45","observation_id":"2cb5a824-c0a3-443a-9659-af964c67444c","resolution":{"observed_at":"2026-08-03T01:58:24.778065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.831865Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.831865Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:d9944c0c36fa5496acc1ceff54a236f307d18ef780cd3683a6d9e7957da92e5f","observation_id":"df0e0675-8033-4cb2-9690-ba9d2d69bd48","resolution":{"observed_at":"2026-08-03T01:58:24.831865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.971000Z","title":"2022.Conversational ai: Dialogue systems, conversational agents, and chatbots","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.971000Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:d80180e05743194d05e0ac71cee0e400d3156d64061bc73d0b5d5e883b5bd7ab","observation_id":"c5953905-38ed-4023-abd8-62ca3f2cdcc7","resolution":{"observed_at":"2026-08-03T01:58:24.971000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.060875Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.060875Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:adca14e320dac1b5da29e75a5a153fdcce9f3858f3112327ecb184806557dbea","observation_id":"5a971130-1e78-444f-802a-210c20cb3444","resolution":{"observed_at":"2026-08-03T01:58:25.060875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15505","last_updated":"2023-10-12T07:55:05Z","snapshot_observed_at":"2026-07-06T16:24:17.829828Z","submitted_at":"2023-09-27T09:13:40Z","title":"Finite Scalar Quantization: VQ-VAE Made Simple","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.15505","snapshot_observed_at":"2026-08-03T01:58:25.125668Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.125668Z"},"links":{"cited_paper":"/paper/2309.15505","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:659bbc01929d437a17d92b6f468eb37187158c30b04ac830f3355ebb9c3ba1f5","observation_id":"f813619f-031a-490c-8b43-b7f25eeb1f6d","resolution":{"observed_at":"2026-08-03T01:58:25.125668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.212574Z","title":null,"venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.212574Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:c1453a81d7b83c7a96049a731b2208e2214422c8d795890606286157c64a900e","observation_id":"9596aed0-4366-48d3-aa18-f4b863be9071","resolution":{"observed_at":"2026-08-03T01:58:25.212574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.274225Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.274225Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:204ebcbdecd74d91009c7f03935bc274d173331a3fc6d2a2bcc55a2def1b9053","observation_id":"f2bed7a0-9916-47e0-9642-245430608ebc","resolution":{"observed_at":"2026-08-03T01:58:25.274225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.344873Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.344873Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:29478fe99dd5e6e1564d5cc698fe17d5f7a24d21b297388e316f887e1e1cad72","observation_id":"fe8408a7-719b-4b94-9a55-aeb1f28999f8","resolution":{"observed_at":"2026-08-03T01:58:25.344873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.390977Z","title":null,"venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.390977Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:2ae1da1b00b7b93c8d10cae656272cbad683159648b8afaa17a12717aa757e44","observation_id":"bc0bf63a-80b6-443f-9553-aa8335fae13e","resolution":{"observed_at":"2026-08-03T01:58:25.390977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.453572Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.453572Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:d08b4d1044986ff156520b6b13789b0f7e50afaef216d302cd64635ffd137c8a","observation_id":"e0f5f0be-5a5b-4169-be5b-23cc5e2a8ff6","resolution":{"observed_at":"2026-08-03T01:58:25.453572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.541344Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.541344Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:596431b2e0c2f906e394607ae3d20629960229b0e6702e0efc252a09c9034eaf","observation_id":"dfa022d7-df70-41de-9d21-8b1e80603f40","resolution":{"observed_at":"2026-08-03T01:58:25.541344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.630635Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.630635Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:60aad14fb36c40bbb13686977a5114990ed37c15a78a39f60fb37a774980c1f0","observation_id":"5eaff0e5-ddd8-46b8-ad63-f7be7a18352d","resolution":{"observed_at":"2026-08-03T01:58:25.630635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.720885Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.720885Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:9ed3c10a1d29f94cba510969cb2cb63f13e076765d8825a023137334961cd59f","observation_id":"bd4c384b-89a3-4a5a-9023-51c708819935","resolution":{"observed_at":"2026-08-03T01:58:25.720885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.785026Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.785026Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:34866f89602fa36c4d8a6f1e9eb1f8393fd185d491dbf9cd8c47610c94bb2c00","observation_id":"970ae424-ccbb-43bb-b882-f69678948f2d","resolution":{"observed_at":"2026-08-03T01:58:25.785026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.17765","last_updated":"2025-09-22T13:26:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-22T13:26:24Z","title":"Qwen3-Omni Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.17765","snapshot_observed_at":"2026-08-03T01:58:25.855309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.855309Z"},"links":{"cited_paper":"/paper/2509.17765","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:5b6516a93ede7fb3260e26ce39b0e1a33de7c8698bcb43067c62369230d9bd00","observation_id":"ed897fd7-6e3c-4127-a5e4-19421ca83663","resolution":{"observed_at":"2026-08-03T01:58:25.855309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:25.938803Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:25.938803Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:7d22450893e532979c3756e3e7b1ab189327e9e0d593a272386cbe739507856d","observation_id":"d86e0a09-d74b-417d-b944-1dd4fe43ff94","resolution":{"observed_at":"2026-08-03T01:58:25.938803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.001410Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.001410Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:cc337f4c5936b7caa6ed49084225d250c9909cadd6bd8c8f9c26e0c3c42feebe","observation_id":"549ebb81-3397-469d-be4f-492469c98441","resolution":{"observed_at":"2026-08-03T01:58:26.001410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.097535Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.097535Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:39239ae0aed2d24eca7a9fb26b777bd17becf2d55c59273cbccd2bac22156255","observation_id":"0f972f44-df81-4562-8a99-5ec31a32148e","resolution":{"observed_at":"2026-08-03T01:58:26.097535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.157040Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.157040Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:4adcfa169cd9405cf21f092358a4c5ba163a94228b18a33d755706f518a3ed1c","observation_id":"b9f9787f-d83b-4ce0-a753-d6dec4bdcad3","resolution":{"observed_at":"2026-08-03T01:58:26.157040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16692","last_updated":"2024-01-23T01:56:57Z","snapshot_observed_at":"2026-08-07T03:22:13.424379Z","submitted_at":"2023-08-31T12:53:09Z","title":"SpeechTokenizer: Unified Speech Tokenizer for Speech Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16692","snapshot_observed_at":"2026-08-03T01:58:26.210946Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.210946Z"},"links":{"cited_paper":"/paper/2308.16692","citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:93789ed011c508956efe50436c5f09ef8e0747b5e8c4ab7b1e2b51dbedd52504","observation_id":"704dae48-2532-4840-ace5-216e89f6dfbc","resolution":{"observed_at":"2026-08-03T01:58:26.210946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.275211Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.275211Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:35916db3b36628a5f4ad9b6c193c3e563e38a0d0df74a63394096d8e85c220ff","observation_id":"ac1f0458-c240-461b-bb6a-31ef825a34dc","resolution":{"observed_at":"2026-08-03T01:58:26.275211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.338340Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.338340Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:da280bafe4cd39cbce7cd75285e27f0ef1442e50fc014aa4c5c1c193d932c62c","observation_id":"80a5dfac-c4b6-4461-b2a7-096c48966777","resolution":{"observed_at":"2026-08-03T01:58:26.338340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.422796Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.422796Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:3bab2c0e3455dc202919414d43e96fe7fafae917552de481be2d65455051d814","observation_id":"953cbca0-3b4b-4a5e-857a-527848ab038e","resolution":{"observed_at":"2026-08-03T01:58:26.422796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.478981Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.478981Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:4478494222c617635d16bb24fa1f288eff76a44a03c1f78a2d4ae671c302a383","observation_id":"0d3912b2-85aa-430d-a135-f9f9fcceb7fc","resolution":{"observed_at":"2026-08-03T01:58:26.478981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.675223Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.675223Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:f41296749f28068205bfb8eada46d75200f947aff0f9068627ec5d66a7d9d2c2","observation_id":"7ab05d96-de4f-44c8-b34c-032b5af45c1d","resolution":{"observed_at":"2026-08-03T01:58:26.675223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.876241Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.876241Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:d4d3854f0a405e49d06382356872b62dc913af11944e755bf31b65feeb4a1b9d","observation_id":"b21ea3c7-74bc-4423-8734-79d982fda5ce","resolution":{"observed_at":"2026-08-03T01:58:26.876241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:24.885164Z","title":"MM ’26, November 10–14, 2026, Rio de Janeiro, Brazil Jia et al","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:24.885164Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:8f8b878f6b0ef447894bc7a3090646e1c614ea4da00b75bbac9b78fde1a50e6f","observation_id":"cc2f75d2-acbc-4a43-a682-4a66a8891e3b","resolution":{"observed_at":"2026-08-03T01:58:24.885164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.578102Z","title":"DATA INTELLIGENCE7, 2 (2025), 527–548","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.578102Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:c65feae6ca1b6535c799f20f18b7b3dd6c5e1ea8e48ae37a89f6d41beb5175df","observation_id":"4224ba59-36a7-4b90-86f1-3ef3fae1ab29","resolution":{"observed_at":"2026-08-03T01:58:26.578102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T01:58:26.766052Z","title":"InProceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-03T01:58:26.766052Z"},"links":{"citing_paper":"/paper/2607.15755"},"observation_digest":"sha256:4a23b185b9ca4ce058b9aedfdf61565d04ab8ff38055d23441a3c2808bd0f986","observation_id":"9ee4c79c-6d68-4882-bb72-83a53b2d5064","resolution":{"observed_at":"2026-08-03T01:58:26.766052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.15755","last_updated":"2026-07-31T04:37:11Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-09T19:02:58.049216Z","submitted_at":"2026-07-17T08:47:58Z","title":"AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":65,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 0 inbound Pith citation observations for arXiv:2607.15755."}