{"as_of":"2026-08-21T08:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:742c2a2828eeed357d8446f9e962b8ed6bcd20c29759333cda96a9a4e7bd36d6","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:00:35.341553Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:00:32.452371Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T17:00:35.528453Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.12015","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12015","snapshot_observed_at":"2026-08-06T17:00:35.528453Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","venue":"cs.SD","work_id":"1d007a0f-de58-4b72-a356-2d460a480fd4","year":2025},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.452371Z"},"links":{"cited_paper":"/paper/2507.12015","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ae3d070b278150d6be37d9b3b32929eef93c9d272bdf8d19f2402e876b865e2c","observation_id":"a5be3642-8335-4aa2-bbcd-e0705651b4ea","resolution":{"observed_at":"2026-08-06T17:00:35.612183Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.12015/citation-record","integrity":"/paper/2507.12015/integrity","json":"/paper/2507.12015/citation-record.json","paper":"/paper/2507.12015"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.12015","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12015","snapshot_observed_at":"2026-08-06T17:00:35.528453Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","venue":"cs.SD","work_id":"1d007a0f-de58-4b72-a356-2d460a480fd4","year":2025},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.452371Z"},"links":{"cited_paper":"/paper/2507.12015","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ae3d070b278150d6be37d9b3b32929eef93c9d272bdf8d19f2402e876b865e2c","observation_id":"a5be3642-8335-4aa2-bbcd-e0705651b4ea","resolution":{"observed_at":"2026-08-06T17:00:35.612183Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:41.254619Z","title":"Overview The overall architecture of EME-TTS is shown in Figure 1a","venue":null,"work_id":"7cbedcf2-95f9-4f7e-b73f-ec0f462e7c9e","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.546382Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:f47635dac1268bac3c0cb66e129003e2533f0a7e03a0ab6f66a8ae71e323d1f3","observation_id":"f3fd56f6-044e-4b73-883b-f093917cbfa0","resolution":{"observed_at":"2026-08-06T17:00:41.294741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:41.019092Z","title":null,"venue":null,"work_id":"ac4025bc-74dc-409d-991f-d937face0b1a","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.720040Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:13882402b19dcc254ff3d12e881fc85e1a68e364f3c7e9bfa70251883954effa","observation_id":"ac11db75-a129-483a-a4e5-ffa6d14a7314","resolution":{"observed_at":"2026-08-06T17:00:41.067248Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.872423Z","title":null,"venue":null,"work_id":"897d9be9-1a8a-426b-9043-d3b0ead865ee","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.761675Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:77a97bc350d7ae3179785bd1b7a0009f1cd7f660b344355466483dab4da57c00","observation_id":"e48b24f3-0364-4ccd-8c9a-fde12479c565","resolution":{"observed_at":"2026-08-06T17:00:40.949596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.702370Z","title":"LQN25F020001), and in part by the Key R&D Program of Zhejiang (2025C01104)","venue":null,"work_id":"1db24f45-9621-413a-b7f4-dd599ca58a15","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.787020Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:9ddeedf5609c707963489f7685f3cd70ed73fa88b1666d08f71d477eab6ac126","observation_id":"e559f109-d637-4a3a-879f-2e350a30d97a","resolution":{"observed_at":"2026-08-06T17:00:40.797255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.641592Z","title":"Naturalspeech: End-to-end text-to- speech synthesis with human-level quality,","venue":null,"work_id":"c1e3a7ca-a054-4aa9-9a40-a37d486caabf","year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.344834Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:76a8f0ce68ace16f47bf4eabbf19ffc3403ae581f0a53ae2ee6c6f6139e9bbd3","observation_id":"6a038c01-4ab0-470b-a154-038b0b632b2f","resolution":{"observed_at":"2026-08-06T17:00:39.711282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.523585Z","title":"Attention is all you need,","venue":null,"work_id":"9cc746de-d7fe-4047-891b-d1f57400f6f6","year":2017},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.844343Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:7ebc953e76a58c5beabd092a654c3d8ccdbfab8f868d4681cf68d8339f476afd","observation_id":"6b0b10a4-0295-460e-83e8-27b08c43b8e7","resolution":{"observed_at":"2026-08-06T17:00:40.623965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.360298Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":"3e5bfef4-a0f0-4c33-8d33-3c34cb2cdd41","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.970639Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:52b9f4b8e9013178b9d3ac7e96f86a3bc2f9cff0a8d983283918cc32ad5f225e","observation_id":"86a78097-fab6-45f8-8473-3cf0aa740f55","resolution":{"observed_at":"2026-08-06T17:00:40.424716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.194686Z","title":"Glow-tts: A genera- tive flow for text-to-speech via monotonic alignment search,","venue":null,"work_id":"8d96ffa8-e20b-4e3c-9531-02bc326e766f","year":2020},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.161679Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ed2a842cf792afeaf72e9f80f0bf652fb68ab716b6b5e352ba24ab9235ba79f3","observation_id":"1bb5d809-8fe0-4f8a-a5d7-4eb3638ff385","resolution":{"observed_at":"2026-08-06T17:00:40.274609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.001426Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"026311a3-34ca-43ad-9300-6397a053e54b","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.220293Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:5c1f3c7182cca3bc57b2233b722857fc4f54ff06cfa6e66e84451c76795f85a8","observation_id":"9b58995a-8d18-4580-8fb0-eef9d33e24f1","resolution":{"observed_at":"2026-08-06T17:00:40.093806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.806308Z","title":"Grad-tts: A diffusion probabilistic model for text-to-speech,","venue":null,"work_id":"e7d7164f-4a4d-481b-9d3b-da987bc352b0","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.285892Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:145d5074d94b8420ec1c258a9e733a2fda06341d7605b26ec7f54381c2039f22","observation_id":"0e9bb990-4e1c-413b-9da5-0f0202d80a17","resolution":{"observed_at":"2026-08-06T17:00:39.897813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.540675Z","title":"Msemotts: Multi-scale emotion transfer, prediction, and control for emotional speech synthesis,","venue":null,"work_id":"cbd92bf4-752f-40cb-a543-a4a1264fdd34","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.800700Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:39aa4c40d187703f4ae85ba2c916ae853cf035dca17faf53e2e04932faa406fd","observation_id":"aa80d873-4dbe-4f85-a15b-fa7f44ee11b1","resolution":{"observed_at":"2026-08-06T17:00:38.642831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.471655Z","title":null,"venue":null,"work_id":"39f29ece-3e6e-4962-a708-3b60f967d566","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.410247Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:82bce282b561e8ba4ed49b9fcb8e75a306d4b61ed07ac7165d1b27cf6b42c706","observation_id":"dddd2a01-6871-4070-9e94-e000cc52b04b","resolution":{"observed_at":"2026-08-06T17:00:39.558958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.267853Z","title":"Exploring transfer learning for low resource emotional tts,","venue":null,"work_id":"cc475391-433f-4873-9a1a-104cba0e0ff1","year":2019},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.487306Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:f6ba3f6eb0b1c0ac1c092c75e0321b28e958fd572a97777215a7c95f3e4f4683","observation_id":"284d4e11-006c-4139-af63-d30e9848ab6e","resolution":{"observed_at":"2026-08-06T17:00:39.338794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.098919Z","title":"Emospeech: guiding fastspeech2 to- wards emotional text to speech,","venue":null,"work_id":"9424c13b-36cc-4549-8c1e-3c0a605cd10c","year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.544569Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:3f6209d6b120f827ecfa7b7a166a9f6dd7e612176134bd6be9ab044b8764f14c","observation_id":"ea7fac22-6097-4904-8b33-7671fa768928","resolution":{"observed_at":"2026-08-06T17:00:39.186750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.945669Z","title":"Emodiff: Intensity con- trollable emotional text-to-speech with soft-label guidance,","venue":null,"work_id":"a2da4601-ce32-46b5-94fe-d98263d84a00","year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.650044Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ef201dd835bd52c7132ff50a59677657d804751170f84e459644151c5928723f","observation_id":"6bc51587-e293-4e72-9edf-f9b106b4b5e8","resolution":{"observed_at":"2026-08-06T17:00:39.018849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.768468Z","title":"Controllable emotion trans- fer for end-to-end speech synthesis,","venue":null,"work_id":"95c291b9-bbdc-4100-9153-18dcdc9745bd","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.743601Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ba9300d4065f8e5790868e53566469daae3620a962f817a3a7f25648d549dd96","observation_id":"be948db1-5421-4a71-8889-1a531c49bb20","resolution":{"observed_at":"2026-08-06T17:00:38.840631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.434088Z","title":"Supervised and unsupervised approaches for controlling narrow lexical focus in sequence-to-sequence speech synthesis,","venue":null,"work_id":"00c04729-1e33-44f1-824c-e6bbe8c08284","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.232683Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:2d2b7b46f17a58c12d5ea0a38abca6bec119e6c22fb43c5174eb2c918d575d25","observation_id":"f947fe3f-0bec-446f-8a11-b72a35a6b2bf","resolution":{"observed_at":"2026-08-06T17:00:37.484831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.362466Z","title":"Daft- exprt: Cross-speaker prosody transfer on any text for expressive speech synthesis,","venue":null,"work_id":"9700795d-0332-48eb-80a1-4ebf332bcd26","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.890240Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:9f860bf143ffdcd25152ccca5c87e0e6989e1fe18273599426fb39203b7148ac","observation_id":"ef779f2c-c39b-4297-9dd7-dec3972d390c","resolution":{"observed_at":"2026-08-06T17:00:38.455301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.119175Z","title":"The acoustics of word stress in en- glish as a function of stress level and speaking style,","venue":null,"work_id":"5aa02811-b276-40cb-9b6e-85556783c128","year":2015},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.967567Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:78b89d9b26cef84bd88814fc4146ab6d6e33b9e9f7e88e2f248009bda4de3314","observation_id":"be7b3e6e-a423-4945-8658-42504f261aa6","resolution":{"observed_at":"2026-08-06T17:00:38.272357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.932748Z","title":"The acoustics of lexical stress in italian as a func- tion of stress level and speaking style,","venue":null,"work_id":"f19cbe53-1da0-4077-9e8b-c15dc764f002","year":2016},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.027610Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:60d792b22078c9cbe082e81c0de96258171e681c915cb69cfb30e502d89218ec","observation_id":"c911473b-ea8e-4858-b756-d0855fd19cd0","resolution":{"observed_at":"2026-08-06T17:00:38.036080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.782074Z","title":"Lex- ical stress perception as a function of acoustic properties and the native language of the listener,","venue":null,"work_id":"f175b854-fb61-4460-93e0-11895536ca02","year":2020},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.084896Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:a61ca5b469419024617bf1292deab523108c6895ec59ee82bf2329bc7869d182","observation_id":"a2e2b0c8-7fce-49fa-8e04-6ea76cd34714","resolution":{"observed_at":"2026-08-06T17:00:37.842467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.574689Z","title":"Emphatic speech generation with conditioned input layer and bidirectional lstms for expressive speech synthesis,","venue":null,"work_id":"3451eed2-fe38-4d8f-97a1-94940f2ac180","year":2018},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.167174Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:01729987c962e68df8f688f2d70805f4f23ab2de2da3cf75a1dcbd409c690e69","observation_id":"1fca9010-44b4-40c8-89b5-f1aab7da3eb5","resolution":{"observed_at":"2026-08-06T17:00:37.669032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:41.126409Z","title":null,"venue":null,"work_id":"b0ffc8d8-920f-4882-b535-4998fb9f5b4d","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.662064Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:c9c21e4e1c8bbb1c3aa62b4e81f903dc08aa027b07f0bed636de8492205b3784","observation_id":"79660d22-0c18-4033-a8fb-aab94c464fe3","resolution":{"observed_at":"2026-08-06T17:00:41.173964Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.253014Z","title":"Emphasis control for parallel neural tts,","venue":null,"work_id":"39f63422-f5d9-4259-b757-b25bba736d8a","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.332717Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:b59b95603872d33db53a77ed93519d975bfc18d4dade93bafc35ef8ae3c2faf6","observation_id":"259c828e-80bb-444b-b2a7-34632b3bd53e","resolution":{"observed_at":"2026-08-06T17:00:37.315296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.053796Z","title":"Ee-tts: Emphatic expressive tts with linguistic information,","venue":null,"work_id":"8449d7d5-961e-4506-8c3a-321ecd677147","year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.389567Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:eb309bd821d2435b04844be9186ccbcd3b3a3ebca93353aa8a8fd84f3585a7bb","observation_id":"f819b3fa-7921-42ac-8baf-696418f5d8db","resolution":{"observed_at":"2026-08-06T17:00:37.146568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-14T10:40:26.323157Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-06T17:00:34.450317Z","title":"A survey of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.450317Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:d4725b5b3f1c6dc6a9ca6d1147d4926bad9ea9811a3c00d75618170c57f2c32f","observation_id":"eefe6eab-eba2-49eb-b8e6-393ef89e99f4","resolution":{"observed_at":"2026-08-06T17:00:34.450317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.856706Z","title":"Em- phAssess : a prosodic benchmark on assessing emphasis transfer in speech-to-speech models,","venue":null,"work_id":"7a0fbcb5-d371-4e29-b4ee-a9dc5ad39554","year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.527568Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:18a7f2c541201d11f37614934afa043200dec391639b7e324268a0fe6c6bb537","observation_id":"b75f85c4-4b12-4115-8b86-c3b7c34fd809","resolution":{"observed_at":"2026-08-06T17:00:36.941557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:34.632284Z","title":"Emotional voice con- version: Theory, databases and esd,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.632284Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:214549fdafe6f1d21420fcaad6851fd52ca7efa0343362ab18987fe36dcafc84","observation_id":"2d4b70b6-65c2-4e28-b4b3-948dc72dbeb6","resolution":{"observed_at":"2026-08-06T17:00:34.632284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.624359Z","title":"Hierarchical rep- resentation and estimation of prosody using continuous wavelet transform,","venue":null,"work_id":"0cfd8b9e-653b-48f1-a74f-648fefdb40ac","year":2017},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.713790Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ab5e35cb6bbe19684c4022f3683f3b837560ec13499e1e196c8fe83976c45f9c","observation_id":"6e06e26b-7eab-48bf-924f-9882a5c7187b","resolution":{"observed_at":"2026-08-06T17:00:36.727689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.415987Z","title":"Adaspeech 4: Adaptive text to speech in zero-shot scenar- ios,","venue":null,"work_id":"20452868-c54b-4954-aa24-241ca41cfcfa","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.803675Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:1cfaa5736b782a50103119e9ef0ff121656adc39d643b482fe8b2ff84a387443","observation_id":"54819339-7cbf-4ad9-b4d7-a7ce54e7406f","resolution":{"observed_at":"2026-08-06T17:00:36.504645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.234021Z","title":"istftnet: Fast and lightweight mel-spectrogram vocoder incorporating inverse short-time fourier transform,","venue":null,"work_id":"4469679c-9fc9-4da2-a0e5-b46926eaffe0","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.887690Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:1ffcb8c3ea324a7845e635ff8af0252dca98c2db49f79e8026c1f63dca256d20","observation_id":"ea73529f-bfe4-47af-a1af-711f4ba6092e","resolution":{"observed_at":"2026-08-06T17:00:36.326584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.097689Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":"96acfd5b-8cf8-450c-8dbc-dccc0d7752b9","year":2015},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.970839Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:265d74b77abcd14ee3ca3b23f696bf7674c5b50cc2b832f5ff2bd073d7d65b7f","observation_id":"2acf7453-1155-4e6c-b3e4-534d2479583e","resolution":{"observed_at":"2026-08-06T17:00:36.152787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-16T06:25:22.037199Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-06T17:00:35.058089Z","title":"Cosyvoice 2: Scalable stream- ing speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.058089Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:52e57c1a9af80e237bf8e2d4dbf0763e45e6a5e0ebe8b9c496ce11b238a4f47d","observation_id":"9df9ccfb-e94f-4307-b1ac-b3977703d489","resolution":{"observed_at":"2026-08-06T17:00:35.058089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:00:35.146817Z","title":"GPT-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.146817Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:2f8aad0f90c9dee72b2cba1206fed914636edb221b21fd73e7c2ded8a737efd9","observation_id":"d419d07b-6e0d-42ae-aee4-43e7d537a330","resolution":{"observed_at":"2026-08-06T17:00:35.146817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:35.911986Z","title":"emotion2vec: Self-supervised pre-training for speech emotion representation,","venue":null,"work_id":"78a2d100-f07c-47b1-8916-84538949e125","year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.245582Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:6c710ab83d631103dc4aa987eb7a1c1c11b488c701ea0cea7d27fd2d22b62c70","observation_id":"9a5cd7ae-cd45-4b15-a515-db92184589a8","resolution":{"observed_at":"2026-08-06T17:00:36.008196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:35.706854Z","title":"Deep learning based assessment of syn- thetic speech naturalness,","venue":null,"work_id":"55602d7b-acd2-4307-9da9-fcc3727fc06f","year":2020},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.341553Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:64e2d24d8159078a1c322f3bfba5c26add2899bded8c0258a84cf33902dd9370","observation_id":"3c4a8d21-99e2-4594-b225-9a385ed9ef18","resolution":{"observed_at":"2026-08-06T17:00:35.809221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":8,"verified_exact":0,"verified_fuzzy":28},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 1 inbound Pith citation observation for arXiv:2507.12015."}