{"as_of":"2026-08-09T13:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f0c8d3fe2ada3f5af86f805cd4f0858c6e440e0b41c6ba5e299dd108029e6768","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T16:54:32.954638Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.17494/citation-record","integrity":"/paper/2508.17494/integrity","json":"/paper/2508.17494/citation-record.json","paper":"/paper/2508.17494"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:32.857688Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.857688Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:3cb7ea35831b10c49b93de6d34fcdd11f52d98adc76d2b2b9c400fed0c7854f7","observation_id":"b978d3a0-0707-43f3-8fbd-85594b02013c","resolution":{"observed_at":"2026-08-05T16:54:32.857688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"publication/3031648","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.490365Z","title":null,"venue":null,"work_id":"eb1348d6-7aac-4254-80f5-cecc210008b5","year":2001},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.861989Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:29fc2bb02c58ed86b49af1e9b5ee0d0d015377535e9e049920778b0b37895fe1","observation_id":"0ae5ae2b-64e7-4fdc-8c6d-ebe93cfc1ee3","resolution":{"observed_at":"2026-08-05T16:54:33.498059Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02132","last_updated":"2023-07-05T09:20:46Z","snapshot_observed_at":"2026-08-07T12:28:23.278063Z","submitted_at":"2023-07-05T09:20:46Z","title":"Going Retro: Astonishingly Simple Yet Effective Rule-based Prosody Modelling for Speech Synthesis Simulating Emotion Dimensions","version":1},"cited_work":{"arxiv_id":"2307.02132","doi":"10.48550/arxiv.2307.02132","metadata_source":"pith","pith_arxiv_id":"2307.02132","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"Going Retro: Astonishingly Simple Yet Effective Rule-based Prosody Modelling for Speech Synthesis Simulating Emotion Dimensions","venue":"cs.SD","work_id":"bc757daf-ff0d-44ab-9fd5-5cce5a62a890","year":2023},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.865949Z"},"links":{"cited_paper":"/paper/2307.02132","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:b1d4a862058b78f73aa6a52bb97fdabca5cc5cb6e516662bdaf74690a57a98a9","observation_id":"ead51eb0-6ab5-4c28-80bb-5f54ee09bebc","resolution":{"observed_at":"2026-08-05T16:54:33.263637Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/speechprosody.2002-35","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.240795Z","title":null,"venue":null,"work_id":"8d06a8f5-1f30-408d-a412-a20ce9a6d9f6","year":2002},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.870634Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:a8cea3e08d2a81f7de3a7e4c8889cb22c47d36c7e3e149d05f21243f7e751ef2","observation_id":"8c036c58-0879-46f0-b072-d7a946649b3e","resolution":{"observed_at":"2026-08-05T16:54:33.244725Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-08T04:22:31.330560Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"cited_work":{"arxiv_id":"2501.04256","doi":"10.48550/arxiv.2501.04256","metadata_source":"pith","pith_arxiv_id":"2501.04256","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","venue":"cs.SD","work_id":"c319522e-9866-4e10-b77a-70896b4b875c","year":2025},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.874696Z"},"links":{"cited_paper":"/paper/2501.04256","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:e52c6d56053f3adfeb084f7e4e4a47434b4d62d99401b2e402241e2ac00e0cbd","observation_id":"994c00c7-d34c-468b-92ae-d298e02b0200","resolution":{"observed_at":"2026-08-05T16:54:33.231805Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-07-06T13:13:37.143547Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-05T16:54:32.879235Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.879235Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:cd698d74aa2438895a6421a8617137365bc013dc12d7bcd9f3b4d3ffd3abed79","observation_id":"4b6359f2-6a4b-4e50-976b-e496ca09749d","resolution":{"observed_at":"2026-08-05T16:54:32.879235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.03600","last_updated":"2022-08-30T16:07:25Z","snapshot_observed_at":"2026-08-09T01:10:41.785557Z","submitted_at":"2021-11-05T16:37:45Z","title":"Hybrid Spectrogram and Waveform Source Separation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.03600","snapshot_observed_at":"2026-08-05T16:54:32.883827Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.883827Z"},"links":{"cited_paper":"/paper/2111.03600","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:32e8cb25ef9af7345762ed6f3a3463e6486fb51146e76558ff9d10c50b88eaeb","observation_id":"bca15180-a774-4167-a984-830a03d9970c","resolution":{"observed_at":"2026-08-05T16:54:32.883827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-05T16:54:32.888561Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.888561Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:587b7eefe5704f5c282702ceaf51149613b859f159fcd9e2c60c7150b8f64387","observation_id":"bcf35241-1b32-4951-84a1-3ed689e29f51","resolution":{"observed_at":"2026-08-05T16:54:32.888561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.12395","last_updated":"2021-04-26T08:29:29Z","snapshot_observed_at":"2026-08-02T08:16:48.794840Z","submitted_at":"2021-04-26T08:29:29Z","title":"Phrase break prediction with bidirectional encoder representations in Japanese text-to-speech synthesis","version":1},"cited_work":{"arxiv_id":"2104.12395","doi":"10.48550/arxiv.2104.12395","metadata_source":"pith","pith_arxiv_id":"2104.12395","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"Phrase break prediction with bidirectional encoder representations in Japanese text-to-speech synthesis","venue":"eess.AS","work_id":"3c69c20b-f711-48da-ad53-1184b14c59da","year":2021},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.892650Z"},"links":{"cited_paper":"/paper/2104.12395","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:5ee44fb27664217393d873a61614811ecc613d53efe1220966b1a74a759c50a8","observation_id":"bee1bfd3-bf6b-47fd-909b-2dd551d0112b","resolution":{"observed_at":"2026-08-05T16:54:33.185706Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2011.02252","last_updated":"2020-11-04T12:20:21Z","snapshot_observed_at":"2026-07-06T10:11:36.210688Z","submitted_at":"2020-11-04T12:20:21Z","title":"Prosodic Representation Learning and Contextual Sampling for Neural Text-to-Speech","version":1},"cited_work":{"arxiv_id":"2011.02252","doi":"10.48550/arxiv.2011.02252","metadata_source":"pith","pith_arxiv_id":"2011.02252","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"Prosodic Representation Learning and Contextual Sampling for Neural Text-to-Speech","venue":"eess.AS","work_id":"7f0afa5c-e585-4563-b04f-74a9f0075a02","year":2020},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.896667Z"},"links":{"cited_paper":"/paper/2011.02252","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:c0cb3ba756f82ebf88c104b3714230b8e12e2a08488867e6269ec43648b1cb4d","observation_id":"b423b86d-489e-4a21-986a-3419d2832811","resolution":{"observed_at":"2026-08-05T16:54:33.166980Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.21437/interspeech.2024-715","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.143781Z","title":null,"venue":null,"work_id":"a25f2b39-bfa8-43e1-aa7a-0db3add8356e","year":2024},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.900528Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:12ae2a2d673d3b1d095adc95f933d3f0de8de44d3576fa4c6905f88f5a60b64e","observation_id":"0627712f-f218-486f-a1a0-70d2193989b6","resolution":{"observed_at":"2026-08-05T16:54:33.147899Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.09577","last_updated":"2019-09-14T03:51:46Z","snapshot_observed_at":"2026-08-03T18:17:55.398298Z","submitted_at":"2019-09-14T03:51:46Z","title":"NeMo: a toolkit for building AI applications using Neural Modules","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.09577","snapshot_observed_at":"2026-08-05T16:54:32.906091Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.906091Z"},"links":{"cited_paper":"/paper/1909.09577","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:3707dfda3b47f6570b71454891f220e6a680ec212cf5abbebd50268cdaa06e5b","observation_id":"4be1c744-4cb7-4e54-8d34-891e26580e3d","resolution":{"observed_at":"2026-08-05T16:54:32.906091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09524","last_updated":"2024-10-12T13:02:31Z","snapshot_observed_at":"2026-08-08T14:33:06.261733Z","submitted_at":"2024-10-12T13:02:31Z","title":"Emphasis Rendering for Conversational Text-to-Speech with Multi-modal Multi-scale Context Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.09524","snapshot_observed_at":"2026-08-05T16:54:32.910025Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.910025Z"},"links":{"cited_paper":"/paper/2410.09524","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:5b233432c4d7e06e5b6b17edaaa74d5d26245019f770266efc38d478d98cf05b","observation_id":"efeb0981-086e-489c-959d-87b30bb53901","resolution":{"observed_at":"2026-08-05T16:54:32.910025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:32.913918Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.913918Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:87e96af31e40a61a6cfb3c0af7dcadaea7a139f6696b80f6a8ee64ab69db2ce7","observation_id":"4ca9dec7-82cb-4636-89a4-32c6a833c8a3","resolution":{"observed_at":"2026-08-05T16:54:32.913918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.569902Z","title":null,"venue":null,"work_id":"f0c6cf1e-386b-4fe9-8725-d21cc9985161","year":2024},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.917483Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:a8552906bbae6f4e20b996771dc89474a1c37e150508155dc13e0c8e9152e8d0","observation_id":"ff278dc9-9a7a-4478-a570-a767aab6dffb","resolution":{"observed_at":"2026-08-05T16:54:33.573745Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1515/lingvan-2022-0156","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.095488Z","title":null,"venue":null,"work_id":"3dbe08a3-6d47-4e52-8a13-50dcb9eca147","year":2024},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.920959Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:eb91028b750d6e153088da3a559c3e2440394f2a88032a8e1f5d8ad8b325173f","observation_id":"514dd63d-bec2-4bac-8a54-1a0f88159ddf","resolution":{"observed_at":"2026-08-05T16:54:33.099364Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06930","last_updated":"2025-01-08T01:33:56Z","snapshot_observed_at":"2026-08-05T05:06:28.984455Z","submitted_at":"2023-10-10T18:33:47Z","title":"Prosody Analysis of Audiobooks","version":3},"cited_work":{"arxiv_id":"2310.06930","doi":"10.48550/arxiv.2310.06930","metadata_source":"pith","pith_arxiv_id":"2310.06930","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"Prosody Analysis of Audiobooks","venue":"cs.SD","work_id":"182040fb-1d0e-4e46-8de6-cf725348f5eb","year":2023},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.924492Z"},"links":{"cited_paper":"/paper/2310.06930","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:61f383a0779ce05f37083eaf8f9fa7eaabd5198eebd45bb43b7c811dc9be433c","observation_id":"c825dd94-b101-42ec-8c8c-6938b910506c","resolution":{"observed_at":"2026-08-05T16:54:33.086196Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-05T16:54:32.928212Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.928212Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:d8caccb18f6d15e2382fb07b1ed40dd2dd5fed2f900816d7210975c55f494aff","observation_id":"4fcb95ce-9e70-474f-88c5-80b778e88558","resolution":{"observed_at":"2026-08-05T16:54:32.928212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.conll-1.31","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.050097Z","title":null,"venue":null,"work_id":"fd67e5be-da46-4be5-882e-e8457ed307c7","year":2023},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.932150Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:ea40168920cf6898f8bcc1ce1147f08ab64ef812030243d716b77a53270cdfd4","observation_id":"cdb03e69-3329-4752-b50d-89f047dc8098","resolution":{"observed_at":"2026-08-05T16:54:33.054101Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.03012","last_updated":"2022-03-29T16:12:30Z","snapshot_observed_at":"2026-08-09T04:27:29.229823Z","submitted_at":"2021-10-06T18:45:39Z","title":"Emphasis control for parallel neural TTS","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.03012","snapshot_observed_at":"2026-08-05T16:54:32.935962Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.935962Z"},"links":{"cited_paper":"/paper/2110.03012","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:af1e51f067c32f37ad0e2ddbe27a320656893a6fec620d7ca14f0259e276098e","observation_id":"244cab55-cf72-4fb3-8a67-7b6793bf89c6","resolution":{"observed_at":"2026-08-05T16:54:32.935962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:32.939730Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.939730Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:23f5e82a621538b19e3fed4fb517b0d2b293d8389b5fabefc9e8546338918211","observation_id":"318e4f26-66a4-451f-afb5-f476e97864d6","resolution":{"observed_at":"2026-08-05T16:54:32.939730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.01718","last_updated":"2022-07-04T20:43:41Z","snapshot_observed_at":"2026-07-06T13:27:46.343350Z","submitted_at":"2022-07-04T20:43:41Z","title":"BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model","version":1},"cited_work":{"arxiv_id":"2207.01718","doi":"10.48550/arxiv.2207.01718","metadata_source":"pith","pith_arxiv_id":"2207.01718","snapshot_observed_at":"2026-08-05T18:16:11.560912Z","title":"BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model","venue":"cs.CL","work_id":"2abc115d-85f0-49b0-8b72-70346bc77f8a","year":2022},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.943237Z"},"links":{"cited_paper":"/paper/2207.01718","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:c5cd83b9ebfb61ba4dfa3c636c39c8206a1e71f8fb0ee26ad7b52794a9dd88ce","observation_id":"b7d8a113-a5db-42d9-8560-103b3c1585d8","resolution":{"observed_at":"2026-08-05T16:54:33.026512Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s42979-024-03652-0","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T16:54:33.001152Z","title":null,"venue":null,"work_id":"463b0997-862c-4771-b7f5-7fb06f334bc4","year":2025},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.947035Z"},"links":{"citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:8a63c7545f2290d75f47c1373e6f3f55304cd51c070c747915989c8bb20b5f21","observation_id":"68b0c917-8034-4595-9999-f43a41d639a7","resolution":{"observed_at":"2026-08-05T16:54:33.006297Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.15067","last_updated":"2022-07-01T01:13:10Z","snapshot_observed_at":"2026-07-06T13:26:14.318051Z","submitted_at":"2022-06-30T07:03:01Z","title":"Language Model-Based Emotion Prediction Methods for Emotional Speech Synthesis Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.15067","snapshot_observed_at":"2026-08-05T16:54:32.950609Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.950609Z"},"links":{"cited_paper":"/paper/2206.15067","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:95aaa3a0335198a22575e00de650b85046ea609e595a9b602e7fd9461c469763","observation_id":"db434059-c06d-4a21-827a-24bc78bcb831","resolution":{"observed_at":"2026-08-05T16:54:32.950609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12107","last_updated":"2024-04-14T12:33:07Z","snapshot_observed_at":"2026-07-06T15:30:01.504305Z","submitted_at":"2023-05-20T05:58:56Z","title":"EE-TTS: Emphatic Expressive TTS with Linguistic Information","version":2},"cited_work":{"arxiv_id":"2305.12107","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.12107","snapshot_observed_at":"2026-08-05T16:54:33.277401Z","title":"EE-TTS: Emphatic Expressive TTS with Linguistic Information","venue":"cs.SD","work_id":"3089b31c-4b83-4f14-ba47-944390c2d3a1","year":2023},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.954638Z"},"links":{"cited_paper":"/paper/2305.12107","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:bc134d461eea3c2dc2a96a9ad0ad246e0f37be21d92dadd027b49c4aeac631ff","observation_id":"aef37265-8f40-4ab2-9be3-85b00144c554","resolution":{"observed_at":"2026-08-05T16:54:33.281755Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T13:16:53.635486Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":13,"verified_fuzzy":0},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2508.17494."}