{"as_of":"2026-08-08T02:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2be0753b4de0832a82410799aec29839636e60fee27a1a55174eaf0552678a3e","coverage":[{"denominator":29,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:43:52.529754Z","state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.20945/citation-record","integrity":"/paper/2506.20945/integrity","json":"/paper/2506.20945/citation-record.json","paper":"/paper/2506.20945"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.842383Z","title":"Mega- tts 2: Boosting prompting mechanisms for zero-shot speech synthesis,","venue":null,"work_id":"6f69a68c-90a3-417e-aef2-e03f88367c37","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.228144Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:88be96c07cd03957fb8f39011626084fe203330ed5d74b6eb47c5482a62b2c8e","observation_id":"27adb6b3-b24d-4ae4-8078-4cbcf618dc2c","resolution":{"observed_at":"2026-08-06T22:43:55.933915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-06T22:43:50.265834Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.265834Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:4ce146645973876e1fe04bf1b800512d4fa56c6bc960d13436d1078238869dd7","observation_id":"49a3a70a-a4b9-4624-8064-993c0af597aa","resolution":{"observed_at":"2026-08-06T22:43:50.265834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00750","last_updated":"2024-10-20T14:25:49Z","snapshot_observed_at":"2026-08-01T10:20:49.367508Z","submitted_at":"2024-09-01T15:26:30Z","title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.00750","snapshot_observed_at":"2026-08-06T22:43:50.332197Z","title":"Maskgct: Zero-shot text-to-speech with masked generative codec transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.332197Z"},"links":{"cited_paper":"/paper/2409.00750","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:9d4cd522f955d1cdb6b2f3938a0cc19b7e8d935d3aca060c1784ec0e33796ad7","observation_id":"663c1c04-c3e4-47e7-9303-e95d307ed5f0","resolution":{"observed_at":"2026-08-06T22:43:50.332197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.634364Z","title":"Imaginary voice: Face-styled diffusion model for text-to-speech,","venue":null,"work_id":"051a348d-6c31-4f0c-9be7-9afe024c670b","year":2023},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.379308Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:0f3b34d0c811f6096fa880c540d0e530cec2d32be65f279c43406cbbef09a05e","observation_id":"80bebff5-0883-408a-bb4e-bb46906172b1","resolution":{"observed_at":"2026-08-06T22:43:55.768449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.478421Z","title":"SYNTHE-SEES: Face based text-to-speech for virtual speaker,","venue":null,"work_id":"3573fd3d-bf0c-4564-9dac-d417fc5cb024","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.454168Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:b398f33c028a4a599330a491bbdcb37e5e3345032a454842eff80e9f146edf8f","observation_id":"3bae3fb5-4477-40bf-a466-d1206948ef89","resolution":{"observed_at":"2026-08-06T22:43:55.560254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.313699Z","title":"Face2Speech: Towards multi-speaker text-to-speech synthesis using an embedding vector predicted from a face image.,","venue":null,"work_id":"429a0ed0-7c91-4f90-a9c4-7e7ce77fdeeb","year":2020},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.528613Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:9c8c56213d2c9518b6b89addd00f8e7d4d7d29340ff8d5c480554c87176101b0","observation_id":"921cc365-a7bf-4ae9-aafd-b477a96c2b58","resolution":{"observed_at":"2026-08-06T22:43:55.395195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.157886Z","title":"FVTTS : Face based voice synthesis for text-to-speech,","venue":null,"work_id":"f5785df7-3794-482c-88ce-fa86b250b42b","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.580539Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:4bdda66d9b293ad5da18726e01e0df790f60135d80dd3dfcbfff05aa94ddb6c6","observation_id":"3cf14a4c-5247-4d3f-a3fd-edced867f33d","resolution":{"observed_at":"2026-08-06T22:43:55.239168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.027835Z","title":"Instructtts: Modelling expressive tts in discrete latent space with natural language style prompt,","venue":null,"work_id":"87ed606f-22cf-4b63-9b54-7c32c544093f","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.633123Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:7cdfd975805b9d40f42f0d254174389ba5609f7769bb1a2c758a6a6aa7bf5a5c","observation_id":"a53c088a-a93b-47ee-9a23-c88649a88f76","resolution":{"observed_at":"2026-08-06T22:43:55.088342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.860818Z","title":"Prompttts: Controllable text-to-speech with text descriptions,","venue":null,"work_id":"aa50650d-d6a9-487b-89d0-3b93e313927f","year":2023},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.701372Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:73075bc06f72e62028f7ccf7c64e5c0770ff353bef0904d15b53e368c1ad414c","observation_id":"ffd35756-5c7b-4282-9d92-0eb646be2e2d","resolution":{"observed_at":"2026-08-06T22:43:54.942659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.713808Z","title":"PromptTTS++: Controlling speaker identity in prompt-based text-to- speech using natural language descriptions,","venue":null,"work_id":"f59b1b09-9bf4-4ab3-932d-b80a7f1e03e5","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.766499Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:f01bea535110be4c14f2ef2ed78a466f2d761ae56e38a3453829c214e550105a","observation_id":"3556fb19-c8c4-4690-a30d-2355895035fb","resolution":{"observed_at":"2026-08-06T22:43:54.781073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18398","last_updated":"2025-02-18T21:39:25Z","snapshot_observed_at":"2026-07-06T18:06:49.190172Z","submitted_at":"2024-04-29T03:19:39Z","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18398","snapshot_observed_at":"2026-08-06T22:43:50.830879Z","title":"Mm-tts: A unified framework for multimodal, prompt-induced emotional text-to-speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.830879Z"},"links":{"cited_paper":"/paper/2404.18398","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:68f76ec268cc98be16da756467bf09af318e4a2cc467d5e2136fff0c610b85dd","observation_id":"96b659af-c3d2-464f-83f5-8ed7af6292f7","resolution":{"observed_at":"2026-08-06T22:43:50.830879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.559656Z","title":"MM-TTS: Multi- modal prompt based style transfer for expressive text-to-speech synthe- sis,","venue":null,"work_id":"3dc12fc1-1c2e-43ce-82ad-828bbe1ef589","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.902999Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:9a173fbf29bdd0626eefe20acad3df882acdc86c4324c8d6f4c71d81d061bcf6","observation_id":"6de8182b-399d-45b8-bde7-dc17676d2a97","resolution":{"observed_at":"2026-08-06T22:43:54.631249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.424176Z","title":"Gen- eralized end-to-end loss for speaker verification,","venue":null,"work_id":"63193231-fdc2-4677-876f-133b6badaceb","year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.984883Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:d7a1cf8e1c98d6f6b79bb6c22d485975f028c4c9e76b6e3dc37af989718c8888","observation_id":"9b3f1fe2-a79f-49e2-ac7e-39310830e03f","resolution":{"observed_at":"2026-08-06T22:43:54.479581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.245956Z","title":"Additive margin softmax for face verification,","venue":null,"work_id":"33d1cf62-afcd-49da-b7c9-c81c2b9102bb","year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.073749Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:098a5d54151ee462dde2071fb9bd3b180a1c942cccff30267f5bf0cfaa2cab85","observation_id":"42cd64a7-6d95-4511-bf4f-adf0c4a098f7","resolution":{"observed_at":"2026-08-06T22:43:54.350835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.063896Z","title":"Bridging the gap between object and image-level representations for open-vocabulary detection,","venue":null,"work_id":"2a3efe4c-cfb6-4972-bb40-912343438c3a","year":2022},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.143577Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:872ac0d6668a575ee7031f30100619b9ef9af023cf0d895cb4af2e277eb43f2c","observation_id":"513fb0e2-5d0e-4c9d-980a-02ad682a3cba","resolution":{"observed_at":"2026-08-06T22:43:54.148216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.897398Z","title":"Joint- teaching: Learning to refine knowledge for resource-constrained un- supervised cross-modal retrieval,","venue":null,"work_id":"7b2b6c54-b680-42e0-9d4c-71ebbd6e0da1","year":2021},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.255727Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:1a663a696391ef263518a4aa7b4bfe39f9d303754dc08d87776e37b479a62a58","observation_id":"1a202008-871a-415c-9415-9faaffdcd3d8","resolution":{"observed_at":"2026-08-06T22:43:54.002813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-06T22:43:51.359236Z","title":"Represen- tation learning with contrastive predictive coding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.359236Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:52248ec787545cf6d298226701709afc5d7aaa563b8920f823fa491a0a986256","observation_id":"8de811c5-5e1f-49dc-865e-18f221ead62d","resolution":{"observed_at":"2026-08-06T22:43:51.359236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.713416Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"73bcf64b-1a33-4fbe-a33c-ab55d88c4b83","year":2021},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.427727Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:45855684cec4bf895d0fa64402841f331673d51bf9a797eb4de4c9077234d77d","observation_id":"283108dd-26b0-4501-ae07-6e229d58e159","resolution":{"observed_at":"2026-08-06T22:43:53.818807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.00496","last_updated":"2018-10-28T14:29:46Z","snapshot_observed_at":"2026-07-06T06:58:51.877372Z","submitted_at":"2018-09-03T08:38:34Z","title":"LRS3-TED: a large-scale dataset for visual speech recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.00496","snapshot_observed_at":"2026-08-06T22:43:51.542194Z","title":"Lrs3- ted: a large-scale dataset for visual speech recognition,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.542194Z"},"links":{"cited_paper":"/paper/1809.00496","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:2a314f2c5fb03ce1a3ba2f331c54a16a7510e00653f5822928a95ad7f2a8f200","observation_id":"a29d89ed-059f-4cef-a988-bf5424033794","resolution":{"observed_at":"2026-08-06T22:43:51.542194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.513270Z","title":"Multi- caption text-to-face synthesis: Dataset and algorithm,","venue":null,"work_id":"f2c8e99f-7766-41bd-a1aa-5eb1f26e6fca","year":2021},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.614760Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:1ff7b12c377ca296c2177ed02362d4e4d8a9e3d996fd4a12d476e6906b51df94","observation_id":"4f76282e-dc9d-4e21-a18a-023dd6db6970","resolution":{"observed_at":"2026-08-06T22:43:53.588907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07969","last_updated":"2024-06-12T07:49:21Z","snapshot_observed_at":"2026-07-06T18:29:25.765578Z","submitted_at":"2024-06-12T07:49:21Z","title":"LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07969","snapshot_observed_at":"2026-08-06T22:43:51.707283Z","title":"LibriTTS-p: A corpus with speaking style and speaker identity prompts for text-to-speech and style caption- ing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.707283Z"},"links":{"cited_paper":"/paper/2406.07969","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:91dff0c1f686464f67241efcfe515f07a1650542b0a486b74bd9eae573f0618c","observation_id":"5eb4f155-d80c-43eb-b966-9739821a71fa","resolution":{"observed_at":"2026-08-06T22:43:51.707283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18802","last_updated":"2023-05-30T07:30:21Z","snapshot_observed_at":"2026-07-06T15:35:15.564727Z","submitted_at":"2023-05-30T07:30:21Z","title":"LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18802","snapshot_observed_at":"2026-08-06T22:43:51.837350Z","title":"Libritts-r: A restored multi-speaker text-to-speech corpus,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.837350Z"},"links":{"cited_paper":"/paper/2305.18802","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:d1b7507c54a4da3d1099dcf41d77b4102d70c09948ec03b49952f3475bb027d6","observation_id":"d4b436d8-9b4e-4335-b459-f4b37136a926","resolution":{"observed_at":"2026-08-06T22:43:51.837350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.02882","last_updated":"2019-04-05T06:05:00Z","snapshot_observed_at":"2026-08-07T13:32:24.356336Z","submitted_at":"2019-04-05T06:05:00Z","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.02882","snapshot_observed_at":"2026-08-06T22:43:51.942109Z","title":"Libritts: A corpus derived from librispeech for text-to-speech,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.942109Z"},"links":{"cited_paper":"/paper/1904.02882","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:ce6e0ca9c6cd66cc578d787df7004ef1b14fde00a1f34d94e178a2c683d84faf","observation_id":"1eb94adb-cfed-44c4-8ab4-2efea73e8fff","resolution":{"observed_at":"2026-08-06T22:43:51.942109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.317074Z","title":"Facenet: A unified embedding for face recognition and clustering,","venue":null,"work_id":"59b03eea-707e-4a7a-9fb4-8f1f4f51f197","year":2015},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.028232Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:49418bc2eb779eb3cd44d0177635e6ce558a78879178508ff2c0cb349ee53208","observation_id":"30bb92a6-34a9-4db0-8670-fc1c27a258f0","resolution":{"observed_at":"2026-08-06T22:43:53.414214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.131288Z","title":"Vggface2: A dataset for recognising faces across pose and age,","venue":null,"work_id":"493da65b-b2d1-4f88-8147-9e45f9e6475f","year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.103593Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:5265a8e9d8d4c692a231f25ddbb17135efc0114808dec28c758fee86e747db44","observation_id":"a1b04ad8-8725-496d-a7bc-e9b50f752036","resolution":{"observed_at":"2026-08-06T22:43:53.221017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:52.934755Z","title":"Joint face detection and alignment using multitask cascaded convolutional networks,","venue":null,"work_id":"bfc79064-ab5a-4e55-867d-d17fa82dfadb","year":2016},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.210637Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:cc233e77f8523fd66971fdff5c1cee077a76b95bac4cfadfb38f8d11e7aae0b9","observation_id":"0398b027-d2a8-43c4-b82e-21d5e5777b9b","resolution":{"observed_at":"2026-08-06T22:43:53.040774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.07143","last_updated":"2020-08-10T13:50:24Z","snapshot_observed_at":"2026-08-07T15:21:18.262728Z","submitted_at":"2020-05-14T17:02:15Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.07143","snapshot_observed_at":"2026-08-06T22:43:52.286082Z","title":"Ecapa- tdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.286082Z"},"links":{"cited_paper":"/paper/2005.07143","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:c919093db6fe904b5dd778115ccdecb545eaf200c9acdc8f127e7a3834f2490b","observation_id":"0859996d-b0fe-41e4-8386-aaebfc5501b1","resolution":{"observed_at":"2026-08-06T22:43:52.286082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.05622","last_updated":"2018-06-27T01:49:17Z","snapshot_observed_at":"2026-08-04T08:09:48.365818Z","submitted_at":"2018-06-14T15:59:12Z","title":"VoxCeleb2: Deep Speaker Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.05622","snapshot_observed_at":"2026-08-06T22:43:52.418324Z","title":"V oxceleb2: Deep speaker recognition,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.418324Z"},"links":{"cited_paper":"/paper/1806.05622","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:c496edffb52833cf409f7cce3547734063a70d988e7df3c172475e78417cbc89","observation_id":"5e0dcf08-eb5f-4c7c-b3d9-6e454ef07ed8","resolution":{"observed_at":"2026-08-06T22:43:52.418324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:52.754031Z","title":"Explor- ing the limits of transfer learning with a unified text-to-text transformer,","venue":null,"work_id":"f1eab33e-336f-40b9-9585-f1e67f453c67","year":2020},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.529754Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:b912de0f687cf018c73cacbeb8881021f19e4023801ac5c381d81730e7d1efa9","observation_id":"385dc97e-106a-4cb1-b335-b4c575076666","resolution":{"observed_at":"2026-08-06T22:43:52.825995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T15:22:41.629254Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis"},"reference_resolution":{"displayed":29,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":19},"total_outbound_references":29},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 29 of 29 outbound references and 0 inbound Pith citation observations for arXiv:2506.20945."}