{"as_of":"2026-08-08T18:20:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ff8b7838e56516e5a8818dae5d40f5aa70709b46b631470e40241de1926468b8","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:28:48.254853Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:28:44.020833Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T14:28:48.777510Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"cited_work":{"arxiv_id":"2505.20341","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20341","snapshot_observed_at":"2026-08-07T14:28:48.777510Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","venue":"eess.AS","work_id":"023f8373-2b88-46c6-b0d2-b7a19f2dfbb6","year":2025},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.020833Z"},"links":{"cited_paper":"/paper/2505.20341","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:11d28a7acd7dc4e7abe96826604d0391bf48310e353b4b3621a33d0c76a9098c","observation_id":"6f217865-cea5-42ed-87f4-44b0d1568529","resolution":{"observed_at":"2026-08-07T14:28:48.852397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.20341/citation-record","integrity":"/paper/2505.20341/integrity","json":"/paper/2505.20341/citation-record.json","paper":"/paper/2505.20341"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"cited_work":{"arxiv_id":"2505.20341","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.20341","snapshot_observed_at":"2026-08-07T14:28:48.777510Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","venue":"eess.AS","work_id":"023f8373-2b88-46c6-b0d2-b7a19f2dfbb6","year":2025},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.020833Z"},"links":{"cited_paper":"/paper/2505.20341","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:11d28a7acd7dc4e7abe96826604d0391bf48310e353b4b3621a33d0c76a9098c","observation_id":"6f217865-cea5-42ed-87f4-44b0d1568529","resolution":{"observed_at":"2026-08-07T14:28:48.852397Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:54.599650Z","title":"sadness” as an ex- ample): “Now, I will give you a sentence. Please modify only one or two words to change the emotion to sadness. Please output only one modified sentence","venue":null,"work_id":"df0a5818-c392-48d9-ab0a-0cc8cb3b9b37","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.161073Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:76485869217c0fc3293c3f918fe31b89e752182f7b3b19fb087b81e71a86ccc5","observation_id":"8eee5942-3a49-489d-a7e9-4e21042c05b8","resolution":{"observed_at":"2026-08-07T14:28:54.751761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:54.117664Z","title":"The first two modules require pre- training, and then the third module is trained end-to-end","venue":null,"work_id":"42fa3bd7-418d-4518-ab05-91330d79da1b","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.341967Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:ead8a20b208d9488dc95cc35cdfeadce672d7070420c8ee5e02c868132b77bd0","observation_id":"43ed3c79-9812-44be-b80f-5831f9cd1cd9","resolution":{"observed_at":"2026-08-07T14:28:54.248425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:53.869010Z","title":"Experimental Setup We evaluate EmoCorrector on the ECD-TSE","venue":null,"work_id":"7bf97c07-f40b-49f5-ab7c-a21ca3f9a742","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.448642Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:1735f0c904ed116c91a9749524783de0eaccbb2fd0f2fa27e5c768c8d01e008c","observation_id":"e2ce15a0-d0e7-44e1-ac88-97f32f3890ae","resolution":{"observed_at":"2026-08-07T14:28:54.011124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:53.681184Z","title":"Experimental results demonstrate that the proposed framework improves emotional consistency in the edited speech","venue":null,"work_id":"4a54f26b-be43-4f71-ba4f-79b3014b8b8b","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.578272Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:9325277e4f2f8c85b2cd8861ca43b70dd8e052fd4f12fb1b98b22b0e412cc571","observation_id":"135f474c-6533-4f97-aec9-9636001b2759","resolution":{"observed_at":"2026-08-07T14:28:53.764375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:53.459966Z","title":"62206136), the General Program (No","venue":null,"work_id":"aba7abe2-10fe-4895-9a5f-23897f419eb6","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.687416Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:661896a951753511d209939c845fdb4d955afb90dc149d26953428679e8b1f74","observation_id":"7ada997d-43c7-48da-b9e4-be65e03beab1","resolution":{"observed_at":"2026-08-07T14:28:53.558362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:45.477922Z","title":"Emotional voice con- version: Theory, databases and esd,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.477922Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:9743127ae044a4ea585e39483dbec8b3bdae3b5370c5150068f75d5abfacb939","observation_id":"2b4da371-5ea5-4768-b7db-d6dab5a649ce","resolution":{"observed_at":"2026-08-07T14:28:45.477922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:53.205377Z","title":"Fluentspeech: Stutter-oriented automatic speech editing with context-aware diffusion models,","venue":null,"work_id":"f3a9b388-a934-4e23-ae5f-2cfe3b330516","year":2023},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.859707Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:2eb3bd7725b7b491d0a819acef8e29fd15e13183d12bafe47d6042ae67e5298a","observation_id":"5bf0d3c4-5ff9-412c-b02f-b716b75d951d","resolution":{"observed_at":"2026-08-07T14:28:53.302583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:52.934339Z","title":"A3t: Alignment-aware acoustic and text pretraining for speech synthe- sis and editing,","venue":null,"work_id":"005a4cf4-f4fe-43e0-bd79-f9f4ffd2f0af","year":2022},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.980567Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:761fdb318c43fc0c3a463a2d92afd07ede93e0b1e352312022f3ec342e6deb04","observation_id":"a957226c-a261-4f04-90b0-3968a953c34a","resolution":{"observed_at":"2026-08-07T14:28:53.069985Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11725","last_updated":"2023-09-22T02:05:36Z","snapshot_observed_at":"2026-07-06T16:21:33.523806Z","submitted_at":"2023-09-21T01:58:01Z","title":"FluentEditor: Text-based Speech Editing by Considering Acoustic and Prosody Consistency","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.11725","snapshot_observed_at":"2026-08-07T14:28:45.077048Z","title":"Fluenteditor: Text-based speech editing by considering acoustic and prosody consistency,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.077048Z"},"links":{"cited_paper":"/paper/2309.11725","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:ea9f436afdd4e4c401e77c3e5064f4db4b83b74350e194308fdb86bd8827a9a0","observation_id":"0b8cb2be-9168-4c7a-87d8-d7fd5af52ee8","resolution":{"observed_at":"2026-08-07T14:28:45.077048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:52.731881Z","title":"Speechx: Neural codec language model as a versatile speech transformer,","venue":null,"work_id":"305a1632-8da2-49fa-a03e-7dc9ed7f2ec9","year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.176163Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:be2efff112bde9cfdd9dba6e42b99ba96d1e6c4b5f2e1671f616ab3332838e29","observation_id":"0a4f1427-3dea-4a21-bdb3-d8564011123e","resolution":{"observed_at":"2026-08-07T14:28:52.829602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:52.480041Z","title":"V oice- craft: Zero-shot speech editing and text-to-speech in the wild,","venue":null,"work_id":"ed1afe7c-f0b7-4f33-89c7-a92657979ff1","year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.291603Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:dfe3b42370fc9c1633bb33e8b7a12d3ede5bb2e72718e4ee9f291d317f94f03a","observation_id":"96ad5d9e-5159-48cf-a12f-b918656bc5c2","resolution":{"observed_at":"2026-08-07T14:28:52.574513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:52.273101Z","title":"Tackling modality heterogeneity with multi-view calibration net- work for multimodal sentiment detection,","venue":null,"work_id":"2a94c60c-a1e0-48c3-91a0-b51f7244054a","year":2023},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.392572Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:c4970a93bf54210a01164dadb90c85adf3795f3c79b6dc5c059d6e599ade0d7d","observation_id":"673ba17a-ef8f-4be3-af29-a216d3511379","resolution":{"observed_at":"2026-08-07T14:28:52.338641Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05746","last_updated":"2024-07-08T08:52:06Z","snapshot_observed_at":"2026-07-06T18:42:51.512751Z","submitted_at":"2024-07-08T08:52:06Z","title":"MSP-Podcast SER Challenge 2024: L'antenne du Ventoux Multimodal Self-Supervised Learning for Speech Emotion Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05746","snapshot_observed_at":"2026-08-07T14:28:46.273638Z","title":"Msp-podcast ser chal- lenge 2024: L’antenne du ventoux multimodal self-supervised learning for speech emotion recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.273638Z"},"links":{"cited_paper":"/paper/2407.05746","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:cd2be686a7e08254ffe43939855cd24ae082a62579c01001d4ce3484c8bb00bb","observation_id":"31dd990b-f13a-4ad6-8b46-7f45de78aaaa","resolution":{"observed_at":"2026-08-07T14:28:46.273638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:52.036282Z","title":"Decoupling speaker-independent emotions for voice conversion via source-filter networks,","venue":null,"work_id":"001ba268-f785-4ebd-835e-cad86dca9c76","year":2023},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.573461Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:325b5b9f3ae70dc6438a7eeacff737517f1241b29c9394fcf051c68ece90be50","observation_id":"4c16ec55-e528-412b-b2bc-b0758292b0cf","resolution":{"observed_at":"2026-08-07T14:28:52.135108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:51.790560Z","title":"Retrieval- augmented generation for knowledge-intensive nlp tasks,","venue":null,"work_id":"e21b5236-01d6-40d2-84d1-de6c71b6766a","year":2020},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.691045Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:e15e98402320f61d28ac3ddaf17fe778011f1c273747869730688aa772685534","observation_id":"9a8e4392-99f9-488b-8d6f-500b10c82f14","resolution":{"observed_at":"2026-08-07T14:28:51.903067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16021","last_updated":"2023-08-30T13:21:51Z","snapshot_observed_at":"2026-08-05T22:19:48.064737Z","submitted_at":"2023-08-30T13:21:51Z","title":"CALM: Contrastive Cross-modal Speaking Style Modeling for Expressive Text-to-Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2308.16021","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.16021","snapshot_observed_at":"2026-08-07T14:28:48.527479Z","title":"CALM: Contrastive Cross-modal Speaking Style Modeling for Expressive Text-to-Speech Synthesis","venue":"cs.SD","work_id":"7de20318-6453-4f7e-aa92-6862a5977095","year":2023},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.816756Z"},"links":{"cited_paper":"/paper/2308.16021","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:f7869de5874d257a51583a677edfaf568b72ae46f91b55b776dbce1c697a14c1","observation_id":"e2d3ddee-1e00-47c9-a852-ce3304452638","resolution":{"observed_at":"2026-08-07T14:28:48.621297Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:51.484998Z","title":"Cross-speaker emotion disentangling and transfer for end-to-end speech synthe- sis,","venue":null,"work_id":"6d86de3c-9d71-4656-ad51-82e6baf9066d","year":2022},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:45.921557Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:7a681caab5313d422eaa78af5e3d5a8f11b121df91acd2158dd8a975f50777f0","observation_id":"0e07d195-a889-4926-be63-00af7372900b","resolution":{"observed_at":"2026-08-07T14:28:51.659972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:54.369926Z","title":"Ultimately, the Azure system synthesizes speech for 5 speakers, CosyV oice2 synthe- sizes speech for another 5 speakers, and F5-TTS synthesizes speech for an additional 2 speakers","venue":null,"work_id":"2dfe1d87-aec9-4483-b2fb-603fb690353c","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:44.265368Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:ae9e3bf932c1415932b0ab1758e71a64a790de9ae766d783fa5ffca9f76c488a","observation_id":"7ce033de-03a7-4c40-ad50-22df9f5efbfa","resolution":{"observed_at":"2026-08-07T14:28:54.460940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:51.211184Z","title":"Disentangling correlated speaker and noise for speech synthesis via data augmentation and adversarial factoriza- tion,","venue":null,"work_id":"8f22e855-8428-49d7-83c7-f3eba6ad36e8","year":2019},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.042428Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:540490a181ba6a3b6ea9d2f7e7d211849e09a089197cafd62258084894f14f75","observation_id":"2642dd13-49f6-4254-a0bf-594957080dcc","resolution":{"observed_at":"2026-08-07T14:28:51.328227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:50.942464Z","title":"Seen and unseen emo- tional style transfer for voice conversion with a new emotional speech dataset,","venue":null,"work_id":"66f36a8e-11f4-41ea-94fe-74f966cddb98","year":2021},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.173639Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:7c3824e38a355c86946cce6cd19be98dfb763be394fd068a2bc59b47a3fb8bd3","observation_id":"10093d43-8a68-498b-bdda-59b7196ec64b","resolution":{"observed_at":"2026-08-07T14:28:51.073704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:46.386970Z","title":"Iemocap: Interactive emotional dyadic motion capture database,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.386970Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:2cae857df427cd34df08f25503fe283e3fc283138134a0187f889d83b1d28d4d","observation_id":"d76e586a-0303-4c6d-9aeb-3a27757aee85","resolution":{"observed_at":"2026-08-07T14:28:46.386970Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:50.695374Z","title":"Azure speech studio,","venue":null,"work_id":"e2557aae-4e98-406c-8429-1ab414392554","year":null},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.491925Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:ecd74789aa50f3e662a514d34255e75761ae471ce3bcc8c098a407ce5da98b6d","observation_id":"68eb2da2-3e00-45b8-b83c-4b10dff8cee3","resolution":{"observed_at":"2026-08-07T14:28:50.841506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-07T06:02:40.568271Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-07T14:28:46.620609Z","title":"Cosyvoice 2: Scalable stream- ing speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.620609Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:1bad9979ffa00a9d82c01eba83a941a990a2249f3647374f16b72455a1269dc0","observation_id":"25852870-80ad-436a-ad82-6a8eac9a38a9","resolution":{"observed_at":"2026-08-07T14:28:46.620609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06885","last_updated":"2025-05-20T15:53:10Z","snapshot_observed_at":"2026-08-01T23:52:25.071174Z","submitted_at":"2024-10-09T13:46:34Z","title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06885","snapshot_observed_at":"2026-08-07T14:28:46.725703Z","title":"F5-tts: A fairytaler that fakes fluent and faithful speech with flow matching,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.725703Z"},"links":{"cited_paper":"/paper/2410.06885","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:b5403f11af38999a011087a71ad006b28bf7a755f57aa257e948e52bf2cc7a0c","observation_id":"c41608d8-a453-4986-893b-393c8db9e335","resolution":{"observed_at":"2026-08-07T14:28:46.725703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:50.426595Z","title":"Mead: A large-scale audio-visual dataset for emotional talking-face generation,","venue":null,"work_id":"b2652e6b-d5b6-449f-9000-2acaab10ac2b","year":2020},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.821401Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:4c2dd77f81b87ccb5e1f2e4939cb041b422434a445c6f16afc05eb63e370ab25","observation_id":"4013cdb8-27d9-47ae-ae3c-7eec84735a0c","resolution":{"observed_at":"2026-08-07T14:28:50.578718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:46.937080Z","title":"Clap learning audio concepts from natural language supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:46.937080Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:6f055f68e4ee42fd04df3f9c2d258834ea735ea129bd61fcdaba646485ccfab6","observation_id":"778845df-7422-4fb4-91fb-f946b3164c71","resolution":{"observed_at":"2026-08-07T14:28:46.937080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1907.11692","last_updated":"2019-07-26T17:48:29Z","snapshot_observed_at":"2026-07-31T22:31:37.910868Z","submitted_at":"2019-07-26T17:48:29Z","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.11692","snapshot_observed_at":"2026-08-07T14:28:47.071239Z","title":"Roberta: A robustly optimized bert pretraining ap- proach,","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.071239Z"},"links":{"cited_paper":"/paper/1907.11692","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:cb16529a7943b31345390d8188f0fa3c9768fcd6e93996f88c1bc453510c8d1c","observation_id":"1d26394d-f402-4627-92fc-1e5aaf6f69b8","resolution":{"observed_at":"2026-08-07T14:28:47.071239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:50.130241Z","title":"emotion2vec: Self-supervised pre-training for speech emotion representation","venue":null,"work_id":"d337becd-c2f6-46dd-b1ba-caecf0984011","year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.187645Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:c943547b75f9346d301c248bab8a875676bc7d26b148e44301dfa904dc87b511","observation_id":"6594f6f7-d31f-4e70-84cd-8276353c998b","resolution":{"observed_at":"2026-08-07T14:28:50.256556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:49.915189Z","title":"Generspeech: Towards style transfer for generalizable out-of-domain text-to- speech,","venue":null,"work_id":"17f59175-e992-4acd-8fc9-c542a9483bb6","year":2022},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.306630Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:4ba431777afa4f551e2d2819df7b86e22354a002f26c4186d0c680431231bfbe","observation_id":"e872d0b9-17ac-46a6-bac7-d619c8bda871","resolution":{"observed_at":"2026-08-07T14:28:50.029173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:47.452788Z","title":"Montreal forced aligner: Trainable text-speech align- ment using kaldi","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.452788Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:3c94583264e5c3216ea8ba2aaf8f94447df0a9cd3059d2b297e63a93977b1d45","observation_id":"b207e2eb-9bbc-49bf-b31f-a04100c052dc","resolution":{"observed_at":"2026-08-07T14:28:47.452788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:49.555457Z","title":"Style tokens: Un- supervised style modeling, control and transfer in end-to-end speech synthesis,","venue":null,"work_id":"c8c7e423-b3ab-4636-91af-a870d352f906","year":2018},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.581107Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:5c65ee823e3c259567a6269a2c04119369174a696824da86303e47efb97d39e4","observation_id":"9c9d3d00-fd43-4ea4-a043-9bed53fb2a6a","resolution":{"observed_at":"2026-08-07T14:28:49.785963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:47.720959Z","title":"Hifi-gan: Generative adversarial net- works for efficient and high fidelity speech synthesis,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.720959Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:783207e25bbd15c5f3e0f42967cf1191c0350277711f525163e4de05194250d2","observation_id":"70bae382-b6d5-4cdf-819f-89643bb7203d","resolution":{"observed_at":"2026-08-07T14:28:47.720959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-07T14:28:47.826489Z","title":"Qwen2-audio technical re- port,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.826489Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:80a9b225682be34b99d7fb9f1813dfd3971cefa4cf55c8b4d8607251d62b1989","observation_id":"0f2aecc1-bdad-4bd6-96b2-0e5b86de5ee5","resolution":{"observed_at":"2026-08-07T14:28:47.826489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:49.219939Z","title":"Editspeech: A text based speech editing system using partial in- ference and bidirectional fusion,","venue":null,"work_id":"73b228cd-3ef2-458c-a590-0cfe82f73174","year":2021},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:47.969837Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:372c9fa9df443092d5c5a433c8f93a71d681187cdadd1bf63476cacd0c1ab3fd","observation_id":"443f867f-75c4-4789-91af-1d32735db711","resolution":{"observed_at":"2026-08-07T14:28:49.387912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.00198","last_updated":"2020-10-24T06:37:42Z","snapshot_observed_at":"2026-07-06T08:54:14.623788Z","submitted_at":"2020-02-01T12:36:55Z","title":"Transforming Spectrum and Prosody for Emotional Voice Conversion with Non-Parallel Training Data","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.00198","snapshot_observed_at":"2026-08-07T14:28:48.097024Z","title":"Transforming spectrum and prosody for emotional voice conversion with non-parallel train- ing data,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:48.097024Z"},"links":{"cited_paper":"/paper/2002.00198","citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:dbec6a2784a3db956b11d5f92bcc0bea1e1f94c49ee695ebb7a0ffcd610ce84d","observation_id":"5b8b15e4-92e7-44f9-be3b-f24a4c2ec7f6","resolution":{"observed_at":"2026-08-07T14:28:48.097024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:28:48.953223Z","title":"Converting anyone’s emotion: Towards speaker-independent emotional voice conver- sion,","venue":null,"work_id":"4bd08dd7-d60b-4d1d-8eee-0692d9f6d9ea","year":2020},"citing_paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:28:48.254853Z"},"links":{"citing_paper":"/paper/2505.20341"},"observation_digest":"sha256:34dca3a72d6882b8797c229e8d6ebf8738321de443f2b82e3567934cb4175d64","observation_id":"8900faa3-d864-46b1-b66e-89201e0ed880","resolution":{"observed_at":"2026-08-07T14:28:49.073961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.20341","last_updated":"2025-05-24T16:10:56Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-07T14:23:48.116577Z","submitted_at":"2025-05-24T16:10:56Z","title":"Towards Emotionally Consistent Text-Based Speech Editing: Introducing EmoCorrector and The ECD-TSE Dataset"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":2,"verified_fuzzy":23},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 1 inbound Pith citation observation for arXiv:2505.20341."}