{"as_of":"2026-08-20T21:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a6ed7fc7dec5109adb11a65ca0fad5ad6ac3e634abac918e510e023d4d47427b","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:41:05.512701Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:54:19.134392Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-10T05:30:23.456663Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-10T05:30:23.456663Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04256","snapshot_observed_at":"2026-08-15T17:54:19.134392Z","title":"Drawspeech: Expressive speech synthesis using prosodic sketches as control conditions","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.20091","last_updated":"2025-08-07T23:23:12Z","snapshot_observed_at":"2026-08-20T06:38:54.503781Z","submitted_at":"2025-07-27T00:59:01Z","title":"ProsodyLM: Uncovering the Emerging Prosody Processing Capabilities in Speech Language Models","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-15T17:54:19.134392Z"},"links":{"cited_paper":"/paper/2501.04256","citing_paper":"/paper/2507.20091"},"observation_digest":"sha256:bbf082c69b49cec0caf4189c4eb99d2af26259192d46d26bddd4904a29a06d7b","observation_id":"ccb609e8-ddf5-4482-843d-273d1b8e377a","resolution":{"observed_at":"2026-08-15T17:54:19.134392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"cited_work":{"arxiv_id":"2501.04256","doi":"10.48550/arxiv.2501.04256","metadata_source":"pith","pith_arxiv_id":"2501.04256","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","venue":"cs.SD","work_id":"c319522e-9866-4e10-b77a-70896b4b875c","year":2025},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-18T00:04:44.525561Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.874696Z"},"links":{"cited_paper":"/paper/2501.04256","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:1915fda1ad5e60e2ff50e7c08602be161cb3ed5f27dadadb0e9f6340fb643498","observation_id":"994c00c7-d34c-468b-92ae-d298e02b0200","resolution":{"observed_at":"2026-08-05T16:54:33.231805Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.04256/citation-record","integrity":"/paper/2501.04256/integrity","json":"/paper/2501.04256/citation-record.json","paper":"/paper/2501.04256"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.722064Z","title":"Review: Prosodic patterns in english conversation,","venue":null,"work_id":"cccbf95f-c056-4e71-beb5-d5ab7d6f2844","year":2020},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.223811Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:6c6b13218a24c7782b86bd5541f453cdb70429e8dc7a2972a85d0e46e8d7fc35","observation_id":"7ed7a4e9-b270-46b6-9c38-b4da6b45fb67","resolution":{"observed_at":"2026-08-10T21:41:06.729039Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.698570Z","title":"Supervised and unsu- pervised approaches for controlling narrow lexical focus in sequence- to-sequence speech synthesis,","venue":null,"work_id":"a655589f-3880-4254-9c91-f3955284f845","year":2021},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.236746Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:5bf9f78542497eb46f9de5ba6eb6d5bb344bef0f41d4bc9c78e2df191524ec1f","observation_id":"f96bab2f-df9c-499a-9184-9bc16d7a53aa","resolution":{"observed_at":"2026-08-10T21:41:06.706053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.15561","last_updated":"2021-07-23T12:32:52Z","snapshot_observed_at":"2026-08-16T18:13:35.460831Z","submitted_at":"2021-06-29T16:50:51Z","title":"A Survey on Neural Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.15561","snapshot_observed_at":"2026-08-10T21:41:05.248498Z","title":"A survey on neural speech synthesis,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.248498Z"},"links":{"cited_paper":"/paper/2106.15561","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:be4b40cc11066341fe81a14714a4952c664b37dc75ea534332c00bba605e1020","observation_id":"b2cafa12-d255-46a0-a3ac-fa7e3d8fe149","resolution":{"observed_at":"2026-08-10T21:41:05.248498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.676740Z","title":"NaturalSpeech: End-to-end text-to-speech synthesis with human-level quality,","venue":null,"work_id":"dfd32f5b-cd98-44e7-bc19-1da4bfdb0189","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.254758Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:fd9ee55b4297bc0c7e1c4bccecf0bd365d45b1fdb5dfa168eef39019376457bc","observation_id":"7283c252-0281-4a8c-a366-db276c7200c3","resolution":{"observed_at":"2026-08-10T21:41:06.683491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.658919Z","title":"Emotion rendering for conversational speech synthesis with heterogeneous graph-based context modeling,","venue":null,"work_id":"8a6b6b85-2fa3-45e4-ab94-1e719f6bc85b","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.260233Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:b54d603675da4795698b6da9a19ae15668e961fe526544f9f553c57020839a1e","observation_id":"4258bcaf-114f-4fd7-8e4f-ae7d127ca892","resolution":{"observed_at":"2026-08-10T21:41:06.664604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.638715Z","title":"Fine-grained emotion strength transfer, control and prediction for emotional speech synthesis,","venue":null,"work_id":"acb293a1-0d0a-4365-82c8-d56468085200","year":2021},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.266983Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:6d8783f4c3a2f2cb0872e9397f7c3fa9d0733d972f11cb384c4a96019d24e3e0","observation_id":"c2cb0723-18e9-4108-bb7e-81a96bde7300","resolution":{"observed_at":"2026-08-10T21:41:06.644394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.03204","last_updated":"2024-05-19T21:34:28Z","snapshot_observed_at":"2026-08-16T14:03:44.923690Z","submitted_at":"2024-04-04T05:15:07Z","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.03204","snapshot_observed_at":"2026-08-10T21:41:05.274200Z","title":"RALL-E: Robust codec language modeling with chain-of-thought prompting for text-to-speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.274200Z"},"links":{"cited_paper":"/paper/2404.03204","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:0f1fba5ec60f701d2fe946281fc98f694271c61de89794654346544f9db0781e","observation_id":"d3dadb15-99c4-42ff-807b-0203a4de419d","resolution":{"observed_at":"2026-08-10T21:41:05.274200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.615747Z","title":"Towards multi-scale style control for expressive speech synthesis,","venue":null,"work_id":"6a09cd5a-d3d0-4c5c-a058-df160cd806ce","year":2021},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.279726Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:7715bae3ed0e4c82ef88a28c08897e9374c0c6bed5954ca13c2bcafedf1bc598","observation_id":"93bbd24b-ed81-455b-8619-2ab0f40e3eb8","resolution":{"observed_at":"2026-08-10T21:41:06.624385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.595723Z","title":"Contrastive context-speech pretraining for expressive text-to-speech synthesis,","venue":null,"work_id":"2cc23f37-3b63-42fa-a9a0-8f2b44634ad3","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.287278Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:5600302bf57fe5e0d9b60f0ef5ee50ca5dc12cf7d96370cc3c1fc45853600d0e","observation_id":"7903e203-d90b-446d-b6a4-8b2a4e9e3ce8","resolution":{"observed_at":"2026-08-10T21:41:06.602743Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.15439","last_updated":"2023-11-20T04:31:13Z","snapshot_observed_at":"2026-08-16T16:56:30.218664Z","submitted_at":"2022-05-30T21:34:40Z","title":"StyleTTS: A Style-Based Generative Model for Natural and Diverse Text-to-Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.15439","snapshot_observed_at":"2026-08-10T21:41:05.292855Z","title":"StyleTTS: A style-based generative model for natural and diverse text-to-speech synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.292855Z"},"links":{"cited_paper":"/paper/2205.15439","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:281ad3ba5c2abb4178fe49284886f733eef01f058ac2d9c3b2b1049affe18016","observation_id":"983c872f-43f9-4b2c-a6a1-4569aa63bad7","resolution":{"observed_at":"2026-08-10T21:41:05.292855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.575334Z","title":"NaturalSpeech 3: Zero-shot speech synthesis with factorized codec and diffusion models,","venue":null,"work_id":"70548094-3abf-4e9e-9225-ec1efe887df4","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.299029Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:4a92ead994d91ae6430de93f6afd687610b21a44d65711372619111f97b2d2ec","observation_id":"ef1b36aa-1cd8-4351-9241-e36eafef24c4","resolution":{"observed_at":"2026-08-10T21:41:06.582016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.556594Z","title":"NaturalSpeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers,","venue":null,"work_id":"2297a45f-b29e-43f0-aace-bd8579d2003c","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.304869Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:573a64af820550b769f5a54eb0c556727482501f0a498a7c8640c37aab8ec39e","observation_id":"1c759169-0b6b-42be-bf31-d0ab2ceebe71","resolution":{"observed_at":"2026-08-10T21:41:06.562966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.534067Z","title":"Principal style components: Expressive style control and cross-speaker transfer in neural tts,","venue":null,"work_id":"cdf67bcb-6c4b-40dc-a820-b7ca36a02ea7","year":2020},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.312897Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:260069742cfd7ad7001ab14212076f5095471a524a5e9e07d9ee3db7698b11c8","observation_id":"9313f290-fd6a-4a42-b03f-88644450ded0","resolution":{"observed_at":"2026-08-10T21:41:06.541323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.507576Z","title":"Multi-speaker expressive speech synthesis via multiple factors decoupling,","venue":null,"work_id":"c66e880f-1941-4d91-a622-35e7abebc140","year":2023},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.324855Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:6d54930f95a376909692e84334afdf37ab96629b22c35e7fae7e36926ae09315","observation_id":"c22c75b0-5fe7-462d-892f-3dfede9a74fc","resolution":{"observed_at":"2026-08-10T21:41:06.514266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05370","last_updated":"2024-06-17T04:39:08Z","snapshot_observed_at":"2026-08-19T20:01:24.492737Z","submitted_at":"2024-06-08T06:31:03Z","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05370","snapshot_observed_at":"2026-08-10T21:41:05.331488Z","title":"V ALL-E 2: Neural codec language models are human parity zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.331488Z"},"links":{"cited_paper":"/paper/2406.05370","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:918b959ded0c085520d186fac27c5767526923a2555abf2d64eb9d89149caa83","observation_id":"372752d7-074a-4fb2-8346-55afcaac34c9","resolution":{"observed_at":"2026-08-10T21:41:05.331488Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.486769Z","title":"Controllable speaking styles using a large language model,","venue":null,"work_id":"02aef787-fc6b-464a-bf33-c384ee1607d5","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.339200Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:664900ab30e44495cc6073d58c0838b9e422c899f4656de3e4b78b6e767ea657","observation_id":"e0854797-73ba-41d9-9c83-a67a0b53735a","resolution":{"observed_at":"2026-08-10T21:41:06.492669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-08-15T11:55:32.600679Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-10T21:41:05.346831Z","title":"Seed-TTS: A family of high-quality versatile speech generation models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.346831Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:e91d06535b451b7275ad031a0763c6049609d1365c0545ea43db322517a22831","observation_id":"7a126917-fb4b-442e-9675-03a4fca3f1c2","resolution":{"observed_at":"2026-08-10T21:41:05.346831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.466348Z","title":"InstructTTS: Modelling expressive tts in discrete latent space with natural language style prompt,","venue":null,"work_id":"3da37fdb-9e25-4de1-84fd-cffe62c5d708","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.353554Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:dc070fb77980792ff8ada683b18439df873ff51a5ae1a1e2bbea59914600046d","observation_id":"c26423aa-3e62-4d57-ba67-c72e1909d8b6","resolution":{"observed_at":"2026-08-10T21:41:06.473423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.447112Z","title":"Controlling emotion in text-to-speech with natural language prompts,","venue":null,"work_id":"ee207c74-8f34-4ac6-981e-eb230f3dc528","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.358581Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:c28867a63ff582d19c7ed24ff1743093d5bef455f395d8f502734cc3f2bffa7d","observation_id":"481a9617-b00e-4e79-9c8f-45c7d41fdc71","resolution":{"observed_at":"2026-08-10T21:41:06.453619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.428090Z","title":"PromptStyle: Controllable style transfer for text-to-speech with natural language descriptions,","venue":null,"work_id":"b3f44c87-0e7d-4a3e-b8bb-2da66b013cfc","year":2023},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.364723Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:79f015e78d805e2832ce98fe4937159a7386ba14ac94977ed4c9a5188e960aea","observation_id":"476e4551-b98e-4449-b8b6-9b008cd9de12","resolution":{"observed_at":"2026-08-10T21:41:06.434421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.402853Z","title":"Prompttts: Controllable text-to-speech with text descriptions,","venue":null,"work_id":"c8bcaf3d-6adf-47c2-998f-c28e9465c21c","year":2023},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.370739Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:f4effaee746f2594d3e8b2734ee0609e936fe6247f6e74a71a0cd4f3e674ae6c","observation_id":"76573e23-412f-47db-8df6-1f4499b66754","resolution":{"observed_at":"2026-08-10T21:41:06.410604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.380074Z","title":"UniAudio: Towards universal audio generation with large language models,","venue":null,"work_id":"2b1ce7ee-9b83-47ab-be37-16d0f46e135b","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.378985Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:53af6aebb4752ae3efb92b4d6214919589cde7caa11351634284406b6e534cac","observation_id":"c2c9d3ff-2ea6-4468-8c90-6303649fe0a0","resolution":{"observed_at":"2026-08-10T21:41:06.388153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-10T21:41:05.384890Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.384890Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:e1b964aa6b2478c8b3975b697a63baa99966bc127bb778c7ca96b49c220fce66","observation_id":"ed8b3ecd-9a45-4ff7-82b7-01d72fac5568","resolution":{"observed_at":"2026-08-10T21:41:05.384890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04051","last_updated":"2024-07-11T02:08:35Z","snapshot_observed_at":"2026-08-16T13:36:59.551255Z","submitted_at":"2024-07-04T16:49:02Z","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04051","snapshot_observed_at":"2026-08-10T21:41:05.397681Z","title":"FunAudioLLM: V oice understanding and generation foundation models for natural interaction between humans and llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.397681Z"},"links":{"cited_paper":"/paper/2407.04051","citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:a2ea165f7d4e91bd3a06b96f9f660276971964cd343a4c9d42e3dd5ba8e1a0ee","observation_id":"638742f1-b16e-4155-a992-b3c8f94fff4d","resolution":{"observed_at":"2026-08-10T21:41:05.397681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.344359Z","title":"Towards general-purpose text-instruction- guided voice conversion,","venue":null,"work_id":"81121578-affc-4f5f-b3e6-59ca37ed545a","year":2023},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.404258Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:1a05c00914b1073407212a793d13d01e060ad9b34efb78cd4bfaef8c50ab2b05","observation_id":"e08c34ec-3071-4c82-896e-f14bdbf5f07b","resolution":{"observed_at":"2026-08-10T21:41:06.350652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.309889Z","title":"BERT: Pre- training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"52f83203-7be8-4ce4-8f26-4c2b9f32a50b","year":2019},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.410582Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:427b9fa1a16d15df267459869443470f89e09a63919f0044c2344b398090463a","observation_id":"2b77f819-72de-432a-becf-d49126aaaa9a","resolution":{"observed_at":"2026-08-10T21:41:06.316711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.287353Z","title":"Communicating emotion: The role of prosodic features,","venue":null,"work_id":"1b33932f-927f-4ac7-94ed-cf8bc276cb94","year":1985},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.417589Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:c7ff61e2d97c67f1bcf00acfc0a3375f46d55117be62905e2a3383d184e79ea1","observation_id":"aa2b4d87-8672-4382-9656-7ac88541cc3d","resolution":{"observed_at":"2026-08-10T21:41:06.293761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.260351Z","title":"Speech prosody enhances the neural processing of syntax,","venue":null,"work_id":"1c049956-ca09-40ad-91ea-1a9f14418c9e","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.424812Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:7c82eaad9e94fc38b21be351403b5c71b67eb89089ebe93664db49064ef57405","observation_id":"7554618b-c9bc-463c-b809-0fcbadbacc9c","resolution":{"observed_at":"2026-08-10T21:41:06.267189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.430583Z","title":"Smoothing and differentiation of data by simplified least squares procedures","venue":null,"work_id":null,"year":1964},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.430583Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:74fa6561d0bf6ebbc438e1ad639a756accd9cf075cb0053ee80d9ac4ca37c503","observation_id":"a5a2a896-3376-4c11-8e20-a9bbc32029fe","resolution":{"observed_at":"2026-08-10T21:41:05.430583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.436238Z","title":"Auto-encoding variational bayes,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.436238Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:ab652cb30f39eb7b89689f4752d04ea4d7663c81d720b76d5c958cad55047a0f","observation_id":"1d6c3ab9-bc4b-441e-a2f6-1982750e480d","resolution":{"observed_at":"2026-08-10T21:41:05.436238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.034975Z","title":"Fast- Speech: Fast, robust and controllable text to speech,","venue":null,"work_id":"3ab851da-1439-46c9-936d-8e4c9fbcfab6","year":2019},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.442259Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:0734167d1b8b7205358f6e3136f119d24828b87a97a27ceb2430c91b2ba8b2d8","observation_id":"46f9ed2b-71af-4d09-82cd-56eea750f524","resolution":{"observed_at":"2026-08-10T21:41:06.050228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:06.002878Z","title":"High- resolution image synthesis with latent diffusion models,","venue":null,"work_id":"d48eb515-374f-44c3-86eb-2b023fa8069b","year":2022},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.448117Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:5b70a213615a40e4cb988ab31749e0a8c09aec1387d7e82b57944d5d71288e81","observation_id":"75cdbde0-ff5f-4f59-a8df-b966c3a4c97e","resolution":{"observed_at":"2026-08-10T21:41:06.011224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.978208Z","title":"Denoising diffusion probabilistic models,","venue":null,"work_id":"ed1dc34a-2ae8-4f77-934a-6c09567e6527","year":2020},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.453438Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:d3a2ed8295f040704fe54165c2a321feffbc52c6617d5d5ac5becad0b6eafd4d","observation_id":"d061b293-93ce-45e5-b679-11e770161819","resolution":{"observed_at":"2026-08-10T21:41:05.985105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.952356Z","title":"AudioLDM: Text-to-audio generation with latent dif- fusion models,","venue":null,"work_id":"2e62d3fa-a27b-47d0-ba33-158c98f9d31f","year":2023},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.458663Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:7330387a2595f0731e84cc8ee52650673869c5141ca722accaa6c3863185a832","observation_id":"b3aabb62-e90f-4bb5-8158-991c067ce2e8","resolution":{"observed_at":"2026-08-10T21:41:05.961826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.463927Z","title":"The lj speech dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.463927Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:874bf8537b4468c0a3242355d288c9819eb1646498b2828467faed7b5ed7989c","observation_id":"50dc9f95-e88f-45da-990f-05ff0f797a25","resolution":{"observed_at":"2026-08-10T21:41:05.463927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.470695Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.470695Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:4814dddd18cac8dced0a99167a073cca6b25d6e4701d8f80109d8dce51fcfe6d","observation_id":"ff91d98f-0b32-45a2-a232-83d0e1c67fe4","resolution":{"observed_at":"2026-08-10T21:41:05.470695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.883886Z","title":"AudioLDM 2: Learning holistic audio generation with self-supervised pretraining,","venue":null,"work_id":"114a5599-fb67-428a-b156-053cc9f54613","year":2024},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.480700Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:832e6cf4c6da5faf27301dd7372633d150e413eb5a7b43b8d8fea9480c7268e8","observation_id":"6ca0b6d7-7058-4eec-aaea-0c2072a50f4c","resolution":{"observed_at":"2026-08-10T21:41:05.892570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.857306Z","title":"HiFi-GAN: Generative adversarial net- works for efficient and high fidelity speech synthesis,","venue":null,"work_id":"f5978082-6dcc-48fc-a2f3-35758cb452a6","year":2020},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.487576Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:08f4f807a07eabc3d07aeb9cfe4796c89ba361bc7baa62f81ed5937a033f3c78","observation_id":"e447c0f9-651a-46fa-b0aa-b804df390b0d","resolution":{"observed_at":"2026-08-10T21:41:05.863672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.834027Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":"c3c1630f-ec88-4068-ae77-bf517a9877ea","year":2015},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.493299Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:59c3e29fe1b81c54cadae459c9a3f3690a3600fe4872b7ed7cb5894b55154e81","observation_id":"bca90c41-3ba0-475a-b9d9-ef63d8bb8fe8","resolution":{"observed_at":"2026-08-10T21:41:05.842468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.804091Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":"09f82e72-2548-4191-8e2b-3bdf83334469","year":2019},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.498808Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:ed3e7c479248eee11b5053aec2c19d3a58d9207763716eb943c7e09e112e87ce","observation_id":"05d7924e-c7c0-4a1e-b8de-2c60cdcff6d4","resolution":{"observed_at":"2026-08-10T21:41:05.811226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.780521Z","title":"FastSpeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":"ba560d68-35b4-4f76-80a8-8bd967d26105","year":2021},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.504092Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:3fdbca3424497052203e0c59bd3aafadf45f71dacb45b320fe6efd43c8d1e36b","observation_id":"14cd95dc-0adb-4123-9c0c-2de2677903ae","resolution":{"observed_at":"2026-08-10T21:41:05.787324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:41:05.752571Z","title":"Individual comparisons by ranking methods,","venue":null,"work_id":"9343de50-daa8-4b1f-8bb9-878c887f7bc8","year":1992},"citing_paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T21:41:05.512701Z"},"links":{"citing_paper":"/paper/2501.04256"},"observation_digest":"sha256:872bd3fe3ff73b6097de70caa9f1b958c25a3c4376820db794ec5c95eb1dbd2c","observation_id":"7da0bfd6-2b54-4525-9bab-93cde587c03f","resolution":{"observed_at":"2026-08-10T21:41:05.760813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.04256","last_updated":"2025-01-08T03:47:54Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-10T21:35:22.314779Z","submitted_at":"2025-01-08T03:47:54Z","title":"DrawSpeech: Expressive Speech Synthesis Using Prosodic Sketches as Control Conditions"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":31},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 2 inbound Pith citation observations for arXiv:2501.04256."}